Files
platform/seed/tools/ocr/index.ts
T
2026-02-24 16:44:22 +00:00

113 lines
3.5 KiB
TypeScript

import { readFileSync, existsSync } from 'node:fs';
import { extname } from 'node:path';
type OcrConfig = {
url: string;
model: string;
api_key?: string;
};
function getOcrConfig(): OcrConfig | null {
try {
const raw = process.env.OFFICER_RESOURCES;
if (!raw) return null;
const resources = JSON.parse(raw) as Record<string, Record<string, string>>;
const ocr = resources['optical-character-recognition'];
if (!ocr?.url) return null;
return { url: ocr.url, model: ocr.model ?? '', api_key: ocr.api_key };
} catch {
return null;
}
}
const DEFAULT_PROMPT = [
'You are an OCR assistant. Extract meaningful text content from images.',
'Rules:',
'- Output ONLY the extracted text, no commentary or explanations.',
'- For documents, articles, books: preserve the original text, paragraphs, and structure as markdown.',
'- For tables: use markdown table format.',
'- For code/terminal screenshots: use fenced code blocks.',
'- For handwritten text: do your best to transcribe accurately.',
'- For mixed content: use appropriate formatting for each section.',
].join('\n');
export async function execute(
_toolCallId: string,
params: { file_path: string; prompt?: string },
) {
const config = getOcrConfig();
if (!config) {
return {
content: [{ type: 'text', text: 'OCR is not configured. Set it up in Settings → Resources → Optical Character Recognition.' }],
isError: true,
};
}
const { file_path, prompt } = params;
if (!existsSync(file_path)) {
return {
content: [{ type: 'text', text: `File not found: ${file_path}` }],
isError: true,
};
}
const imageBytes = readFileSync(file_path);
const base64 = imageBytes.toString('base64');
const ext = extname(file_path).replace('.', '').toLowerCase();
const mime = ext === 'jpg' || ext === 'jpeg' ? 'image/jpeg' : ext === 'webp' ? 'image/webp' : ext === 'gif' ? 'image/gif' : `image/${ext || 'png'}`;
const headers: Record<string, string> = { 'Content-Type': 'application/json' };
if (config.api_key) headers['Authorization'] = `Bearer ${config.api_key}`;
const systemPrompt = prompt ? `${DEFAULT_PROMPT}\n\nAdditional instructions: ${prompt}` : DEFAULT_PROMPT;
try {
const res = await fetch(`${config.url.replace(/\/+$/, '')}/v1/chat/completions`, {
method: 'POST',
headers,
body: JSON.stringify({
model: config.model,
messages: [
{ role: 'system', content: systemPrompt },
{
role: 'user',
content: [
{ type: 'image_url', image_url: { url: `data:${mime};base64,${base64}` } },
],
},
],
max_tokens: 4096,
}),
});
if (!res.ok) {
const errorText = await res.text().catch(() => '');
return {
content: [{ type: 'text', text: `OCR API error (${res.status}): ${errorText}` }],
isError: true,
};
}
const json = await res.json() as { choices?: Array<{ message?: { content?: string } }> };
const text = json.choices?.[0]?.message?.content ?? '';
if (!text) {
return {
content: [{ type: 'text', text: 'OCR returned empty result — the image may not contain readable text.' }],
isError: false,
};
}
return {
content: [{ type: 'text', text }],
};
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
return {
content: [{ type: 'text', text: `OCR request failed: ${message}` }],
isError: true,
};
}
}