Add Tesseract OCR fallback to pdf_read
스캔 PDF에서 텍스트 추출 실패 시 자동으로 OCR 실행. PyMuPDF로 300DPI 이미지 렌더링 후 tesseract kor+eng 적용. ocr: true 파라미터로 강제 OCR 가능. method 필드로 추출 방법 표시. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+87
-23
@@ -17,22 +17,70 @@ function getWorkspacePath(args?: any): string {
|
||||
}
|
||||
}
|
||||
|
||||
const OCR_SCRIPT = `
|
||||
import sys, os, subprocess, tempfile
|
||||
try:
|
||||
import fitz
|
||||
except ImportError:
|
||||
print("ERROR:fitz_missing")
|
||||
sys.exit(1)
|
||||
|
||||
pdf_path = sys.argv[1]
|
||||
page_from = int(sys.argv[2]) if len(sys.argv) > 2 else 1
|
||||
page_to = int(sys.argv[3]) if len(sys.argv) > 3 else 0
|
||||
lang = sys.argv[4] if len(sys.argv) > 4 else 'kor+eng'
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
total = doc.page_count
|
||||
end = min(page_to, total) if page_to > 0 else total
|
||||
texts = []
|
||||
|
||||
for i in range(page_from - 1, end):
|
||||
page = doc[i]
|
||||
mat = fitz.Matrix(300/72, 300/72)
|
||||
pix = page.get_pixmap(matrix=mat)
|
||||
tmp = tempfile.NamedTemporaryFile(suffix='.png', delete=False)
|
||||
pix.save(tmp.name)
|
||||
tmp.close()
|
||||
try:
|
||||
r = subprocess.run(
|
||||
['tesseract', tmp.name, 'stdout', '-l', lang, '--psm', '6'],
|
||||
capture_output=True, text=True, timeout=120
|
||||
)
|
||||
texts.append(f'--- Page {i+1} ---\\n{r.stdout.strip()}')
|
||||
finally:
|
||||
os.unlink(tmp.name)
|
||||
|
||||
print('\\n\\n'.join(texts))
|
||||
`;
|
||||
|
||||
async function runOcr(resolved: string, pageFrom: number, pageTo: number): Promise<string> {
|
||||
const { stdout } = await execFileAsync(
|
||||
'python3', ['-c', OCR_SCRIPT, resolved, String(pageFrom), String(pageTo), 'kor+eng'],
|
||||
{ maxBuffer: 20 * 1024 * 1024, timeout: 300_000 }
|
||||
);
|
||||
if (stdout.startsWith('ERROR:fitz_missing')) throw new Error('PyMuPDF(fitz) not installed');
|
||||
return stdout.trim();
|
||||
}
|
||||
|
||||
export const pdfReadTool = {
|
||||
name: 'pdf_read',
|
||||
description: 'Extract text from a PDF file. Returns full text content, optionally limited to a page range. Ideal for reading research papers, reports, and documents.',
|
||||
description: 'Extract text from a PDF file. For text-based PDFs uses pdftotext; for scanned/image PDFs automatically falls back to Tesseract OCR (kor+eng). Returns full text, optionally limited to a page range.',
|
||||
schema: {
|
||||
path: 'Path to the PDF file (absolute, or relative to workspace)',
|
||||
page_from: 'First page to extract, 1-indexed (default: 1)',
|
||||
page_to: 'Last page to extract, inclusive (default: last page)',
|
||||
max_chars: 'Maximum characters to return (default: 20000, max: 100000)',
|
||||
ocr: 'Force OCR even if text is extractable: true | false (default: auto)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the PDF file' },
|
||||
page_from: { type: 'number', description: 'First page (1-indexed)' },
|
||||
page_to: { type: 'number', description: 'Last page (inclusive)' },
|
||||
max_chars: { type: 'number', description: 'Max chars to return (default: 20000)' },
|
||||
path: { type: 'string', description: 'Path to the PDF file' },
|
||||
page_from: { type: 'number', description: 'First page (1-indexed)' },
|
||||
page_to: { type: 'number', description: 'Last page (inclusive)' },
|
||||
max_chars: { type: 'number', description: 'Max chars to return (default: 20000)' },
|
||||
ocr: { type: 'boolean', description: 'Force OCR (default: auto-detect)' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
@@ -51,27 +99,43 @@ export const pdfReadTool = {
|
||||
return { success: false, error: 'File must have a .pdf extension' };
|
||||
}
|
||||
|
||||
const maxChars = Math.min(100_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
|
||||
const cmdArgs: string[] = ['-layout', '-enc', 'UTF-8'];
|
||||
if (args?.page_from) cmdArgs.push('-f', String(Math.max(1, Math.floor(Number(args.page_from)))));
|
||||
if (args?.page_to) cmdArgs.push('-l', String(Math.max(1, Math.floor(Number(args.page_to)))));
|
||||
cmdArgs.push(resolved, '-');
|
||||
const maxChars = Math.min(100_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
|
||||
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : 1;
|
||||
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : 0;
|
||||
const forceOcr = args?.ocr === true;
|
||||
|
||||
let stdout: string;
|
||||
try {
|
||||
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
|
||||
stdout = result.stdout;
|
||||
} catch (err: any) {
|
||||
const msg = String(err.message || '');
|
||||
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
|
||||
return { success: false, error: 'PDF is password-protected or encrypted' };
|
||||
let text = '';
|
||||
let method = 'pdftotext';
|
||||
|
||||
if (!forceOcr) {
|
||||
const cmdArgs: string[] = ['-layout', '-enc', 'UTF-8'];
|
||||
if (pageFrom > 1) cmdArgs.push('-f', String(pageFrom));
|
||||
if (pageTo > 0) cmdArgs.push('-l', String(pageTo));
|
||||
cmdArgs.push(resolved, '-');
|
||||
|
||||
try {
|
||||
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
|
||||
text = result.stdout.replace(/\r/g, '').trim();
|
||||
} catch (err: any) {
|
||||
const msg = String(err.message || '');
|
||||
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
|
||||
return { success: false, error: 'PDF is password-protected or encrypted' };
|
||||
}
|
||||
return { success: false, error: `pdftotext failed: ${msg}` };
|
||||
}
|
||||
return { success: false, error: `pdftotext failed: ${msg}` };
|
||||
}
|
||||
|
||||
const text = stdout.replace(/\r/g, '').trim();
|
||||
// Fallback to OCR if no text extracted
|
||||
if (!text) {
|
||||
return { success: false, error: 'No text extracted. The PDF may be scanned (image-based). Try image_read with OCR instead.' };
|
||||
method = 'ocr';
|
||||
try {
|
||||
text = await runOcr(resolved, pageFrom, pageTo);
|
||||
} catch (err: any) {
|
||||
return { success: false, error: `OCR failed: ${err.message}` };
|
||||
}
|
||||
if (!text) {
|
||||
return { success: false, error: 'OCR returned no text. The PDF may be empty or unreadable.' };
|
||||
}
|
||||
}
|
||||
|
||||
const truncated = text.length > maxChars;
|
||||
@@ -80,9 +144,9 @@ export const pdfReadTool = {
|
||||
return {
|
||||
success: true,
|
||||
stdout: truncated
|
||||
? `${output}\n\n[Truncated at ${maxChars.toLocaleString()} chars. Full length: ~${text.length.toLocaleString()} chars]`
|
||||
? `${output}\n\n[Truncated at ${maxChars.toLocaleString()} chars. Full: ~${text.length.toLocaleString()} chars]`
|
||||
: output,
|
||||
data: { chars: text.length, truncated },
|
||||
data: { chars: text.length, truncated, method },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user