Add Tesseract OCR fallback to pdf_read

스캔 PDF에서 텍스트 추출 실패 시 자동으로 OCR 실행.
PyMuPDF로 300DPI 이미지 렌더링 후 tesseract kor+eng 적용.
ocr: true 파라미터로 강제 OCR 가능. method 필드로 추출 방법 표시.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
kim
2026-05-19 13:04:52 +09:00
co-authored by Claude Sonnet 4.6
parent 8135477d12
commit bac5604d5e
+87 -23
View File
@@ -17,22 +17,70 @@ function getWorkspacePath(args?: any): string {
}
}
const OCR_SCRIPT = `
import sys, os, subprocess, tempfile
try:
import fitz
except ImportError:
print("ERROR:fitz_missing")
sys.exit(1)
pdf_path = sys.argv[1]
page_from = int(sys.argv[2]) if len(sys.argv) > 2 else 1
page_to = int(sys.argv[3]) if len(sys.argv) > 3 else 0
lang = sys.argv[4] if len(sys.argv) > 4 else 'kor+eng'
doc = fitz.open(pdf_path)
total = doc.page_count
end = min(page_to, total) if page_to > 0 else total
texts = []
for i in range(page_from - 1, end):
page = doc[i]
mat = fitz.Matrix(300/72, 300/72)
pix = page.get_pixmap(matrix=mat)
tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False)
pix.save(tmp.name)
tmp.close()
try:
r = subprocess.run(
['tesseract', tmp.name, 'stdout', '-l', lang, '--psm', '6'],
capture_output=True, text=True, timeout=120
)
texts.append(f'--- Page {i+1} ---\\n{r.stdout.strip()}')
finally:
os.unlink(tmp.name)
print('\\n\\n'.join(texts))
`;
async function runOcr(resolved: string, pageFrom: number, pageTo: number): Promise<string> {
const { stdout } = await execFileAsync(
'python3', ['-c', OCR_SCRIPT, resolved, String(pageFrom), String(pageTo), 'kor+eng'],
{ maxBuffer: 20 * 1024 * 1024, timeout: 300_000 }
);
if (stdout.startsWith('ERROR:fitz_missing')) throw new Error('PyMuPDF(fitz) not installed');
return stdout.trim();
}
export const pdfReadTool = {
name: 'pdf_read',
description: 'Extract text from a PDF file. Returns full text content, optionally limited to a page range. Ideal for reading research papers, reports, and documents.',
description: 'Extract text from a PDF file. For text-based PDFs uses pdftotext; for scanned/image PDFs automatically falls back to Tesseract OCR (kor+eng). Returns full text, optionally limited to a page range.',
schema: {
path: 'Path to the PDF file (absolute, or relative to workspace)',
page_from: 'First page to extract, 1-indexed (default: 1)',
page_to: 'Last page to extract, inclusive (default: last page)',
max_chars: 'Maximum characters to return (default: 20000, max: 100000)',
ocr: 'Force OCR even if text is extractable: true | false (default: auto)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string', description: 'Path to the PDF file' },
page_from: { type: 'number', description: 'First page (1-indexed)' },
page_to: { type: 'number', description: 'Last page (inclusive)' },
max_chars: { type: 'number', description: 'Max chars to return (default: 20000)' },
path: { type: 'string', description: 'Path to the PDF file' },
page_from: { type: 'number', description: 'First page (1-indexed)' },
page_to: { type: 'number', description: 'Last page (inclusive)' },
max_chars: { type: 'number', description: 'Max chars to return (default: 20000)' },
ocr: { type: 'boolean', description: 'Force OCR (default: auto-detect)' },
},
required: ['path'],
additionalProperties: false,
@@ -51,27 +99,43 @@ export const pdfReadTool = {
return { success: false, error: 'File must have a .pdf extension' };
}
const maxChars = Math.min(100_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
const cmdArgs: string[] = ['-layout', '-enc', 'UTF-8'];
if (args?.page_from) cmdArgs.push('-f', String(Math.max(1, Math.floor(Number(args.page_from)))));
if (args?.page_to) cmdArgs.push('-l', String(Math.max(1, Math.floor(Number(args.page_to)))));
cmdArgs.push(resolved, '-');
const maxChars = Math.min(100_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : 1;
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : 0;
const forceOcr = args?.ocr === true;
let stdout: string;
try {
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
stdout = result.stdout;
} catch (err: any) {
const msg = String(err.message || '');
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
return { success: false, error: 'PDF is password-protected or encrypted' };
let text = '';
let method = 'pdftotext';
if (!forceOcr) {
const cmdArgs: string[] = ['-layout', '-enc', 'UTF-8'];
if (pageFrom > 1) cmdArgs.push('-f', String(pageFrom));
if (pageTo > 0) cmdArgs.push('-l', String(pageTo));
cmdArgs.push(resolved, '-');
try {
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
text = result.stdout.replace(/\r/g, '').trim();
} catch (err: any) {
const msg = String(err.message || '');
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
return { success: false, error: 'PDF is password-protected or encrypted' };
}
return { success: false, error: `pdftotext failed: ${msg}` };
}
return { success: false, error: `pdftotext failed: ${msg}` };
}
const text = stdout.replace(/\r/g, '').trim();
// Fallback to OCR if no text extracted
if (!text) {
return { success: false, error: 'No text extracted. The PDF may be scanned (image-based). Try image_read with OCR instead.' };
method = 'ocr';
try {
text = await runOcr(resolved, pageFrom, pageTo);
} catch (err: any) {
return { success: false, error: `OCR failed: ${err.message}` };
}
if (!text) {
return { success: false, error: 'OCR returned no text. The PDF may be empty or unreadable.' };
}
}
const truncated = text.length > maxChars;
@@ -80,9 +144,9 @@ export const pdfReadTool = {
return {
success: true,
stdout: truncated
? `${output}\n\n[Truncated at ${maxChars.toLocaleString()} chars. Full length: ~${text.length.toLocaleString()} chars]`
? `${output}\n\n[Truncated at ${maxChars.toLocaleString()} chars. Full: ~${text.length.toLocaleString()} chars]`
: output,
data: { chars: text.length, truncated },
data: { chars: text.length, truncated, method },
};
},
};