feat(ocr): fine-tune OCR quality criteria with stricter threshold and head-to-head comparison
- Raise quality acceptance threshold from 65→85 (configurable via TEXT_QUALITY_THRESHOLD) - Reject text with significant issues (excessive_typos, garbage_characters, incoherent_text, fragmented_sentences) even when score is above threshold (configurable via TEXT_QUALITY_SIGNIFICANT_ISSUES) - Add compare_text_quality() for AI-powered head-to-head comparison of original embedded text vs fresh OCR output - Update process_document to pass original text to OCR task for comparison - Update process_with_ocr to run comparison and keep the higher-quality text - Add new settings to settings_service.py metadata - Update docs/ConfigurationGuide.md with new settings - Add comprehensive tests for new threshold and comparison logic Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
@@ -428,19 +428,21 @@ def process_document(
|
||||
|
||||
if not quality_result.is_good_quality:
|
||||
# Poor quality: discard embedded text and re-OCR instead.
|
||||
# Pass the original embedded text so the OCR task can compare
|
||||
# its result against the original and keep the better version.
|
||||
issues_str = ", ".join(quality_result.issues) if quality_result.issues else "unspecified"
|
||||
detail_msg = (
|
||||
f"Text quality check FAILED – score={quality_result.quality_score}/100, "
|
||||
f"source={quality_result.text_source.value}, issues=[{issues_str}].\n"
|
||||
f"AI feedback: {quality_result.feedback}\n"
|
||||
f"Embedded text will be ignored; re-running OCR."
|
||||
f"Embedded text will be compared with fresh OCR output; best version will be used."
|
||||
)
|
||||
logger.warning(f"[{task_id}] {detail_msg}")
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"check_text_quality",
|
||||
"failure",
|
||||
f"Poor quality text (score={quality_result.quality_score}/100); queuing OCR",
|
||||
f"Poor quality text (score={quality_result.quality_score}/100); queuing OCR for comparison",
|
||||
file_id=file_id,
|
||||
detail=detail_msg,
|
||||
)
|
||||
@@ -451,7 +453,7 @@ def process_document(
|
||||
"Queued for OCR (text quality too low)",
|
||||
file_id=file_id,
|
||||
)
|
||||
process_with_ocr.delay(new_filename, file_id)
|
||||
process_with_ocr.delay(new_filename, file_id, extracted_text)
|
||||
return {
|
||||
"file": new_local_path,
|
||||
"status": "Queued for OCR (poor embedded text quality)",
|
||||
|
||||
Reference in New Issue
Block a user