feat(ocr): fine-tune OCR quality criteria with stricter threshold and head-to-head comparison
- Raise quality acceptance threshold from 65→85 (configurable via TEXT_QUALITY_THRESHOLD) - Reject text with significant issues (excessive_typos, garbage_characters, incoherent_text, fragmented_sentences) even when score is above threshold (configurable via TEXT_QUALITY_SIGNIFICANT_ISSUES) - Add compare_text_quality() for AI-powered head-to-head comparison of original embedded text vs fresh OCR output - Update process_document to pass original text to OCR task for comparison - Update process_with_ocr to run comparison and keep the higher-quality text - Add new settings to settings_service.py metadata - Update docs/ConfigurationGuide.md with new settings - Add comprehensive tests for new threshold and comparison logic Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
@@ -288,6 +288,22 @@ class Settings(BaseSettings):
|
||||
"bypass the check. Default: True (enabled)."
|
||||
),
|
||||
)
|
||||
text_quality_threshold: int = Field(
|
||||
default=85,
|
||||
description=(
|
||||
"Minimum quality score (0–100) required to accept embedded PDF text without re-OCR. "
|
||||
"Text scoring below this threshold is discarded and the file is re-processed with OCR. "
|
||||
"Default: 85. The stricter this value, the more files will be re-OCR'd."
|
||||
),
|
||||
)
|
||||
text_quality_significant_issues: Union[List[str], str] = Field(
|
||||
default_factory=lambda: ["excessive_typos", "garbage_characters", "incoherent_text", "fragmented_sentences"],
|
||||
description=(
|
||||
"Comma-separated list of quality issue labels that force OCR re-run even when the quality "
|
||||
"score is above TEXT_QUALITY_THRESHOLD. Any of these issues present in the AI assessment "
|
||||
"will trigger re-OCR. Default: excessive_typos,garbage_characters,incoherent_text,fragmented_sentences"
|
||||
),
|
||||
)
|
||||
|
||||
# Processing step timeout - prevents files from getting stuck in "in_progress" state
|
||||
step_timeout: int = Field(
|
||||
@@ -443,6 +459,18 @@ class Settings(BaseSettings):
|
||||
return []
|
||||
return v
|
||||
|
||||
@field_validator("text_quality_significant_issues", mode="before")
|
||||
@classmethod
|
||||
def parse_text_quality_significant_issues(cls, v: str | list[str]) -> list[str]:
|
||||
"""Parse significant issue labels from comma-separated string or list."""
|
||||
if isinstance(v, str):
|
||||
if "," in v:
|
||||
return [item.strip() for item in v.split(",") if item.strip()]
|
||||
elif v.strip():
|
||||
return [v.strip()]
|
||||
return []
|
||||
return v
|
||||
|
||||
@field_validator("cors_allowed_origins", "cors_allowed_methods", "cors_allowed_headers", mode="before")
|
||||
@classmethod
|
||||
def parse_comma_separated_list(cls, v: str | list[str]) -> list[str]:
|
||||
|
||||
@@ -428,19 +428,21 @@ def process_document(
|
||||
|
||||
if not quality_result.is_good_quality:
|
||||
# Poor quality: discard embedded text and re-OCR instead.
|
||||
# Pass the original embedded text so the OCR task can compare
|
||||
# its result against the original and keep the better version.
|
||||
issues_str = ", ".join(quality_result.issues) if quality_result.issues else "unspecified"
|
||||
detail_msg = (
|
||||
f"Text quality check FAILED – score={quality_result.quality_score}/100, "
|
||||
f"source={quality_result.text_source.value}, issues=[{issues_str}].\n"
|
||||
f"AI feedback: {quality_result.feedback}\n"
|
||||
f"Embedded text will be ignored; re-running OCR."
|
||||
f"Embedded text will be compared with fresh OCR output; best version will be used."
|
||||
)
|
||||
logger.warning(f"[{task_id}] {detail_msg}")
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"check_text_quality",
|
||||
"failure",
|
||||
f"Poor quality text (score={quality_result.quality_score}/100); queuing OCR",
|
||||
f"Poor quality text (score={quality_result.quality_score}/100); queuing OCR for comparison",
|
||||
file_id=file_id,
|
||||
detail=detail_msg,
|
||||
)
|
||||
@@ -451,7 +453,7 @@ def process_document(
|
||||
"Queued for OCR (text quality too low)",
|
||||
file_id=file_id,
|
||||
)
|
||||
process_with_ocr.delay(new_filename, file_id)
|
||||
process_with_ocr.delay(new_filename, file_id, extracted_text)
|
||||
return {
|
||||
"file": new_local_path,
|
||||
"status": "Queued for OCR (poor embedded text quality)",
|
||||
|
||||
@@ -8,7 +8,10 @@ task with a multi-engine OCR pipeline that:
|
||||
2. Merges/cross-checks the results using the configured AI model when more
|
||||
than one provider is active (see ``OCR_MERGE_STRATEGY``).
|
||||
3. Writes the best searchable PDF back to the working directory.
|
||||
4. Hands off to the page-rotation and metadata-extraction pipeline exactly as
|
||||
4. Optionally compares the OCR output against the original embedded text
|
||||
(passed as *original_text*) using a head-to-head AI review and keeps the
|
||||
higher-quality text for downstream processing.
|
||||
5. Hands off to the page-rotation and metadata-extraction pipeline exactly as
|
||||
the legacy Azure task did.
|
||||
"""
|
||||
|
||||
@@ -22,20 +25,26 @@ from app.tasks.retry_config import BaseTaskWithRetry
|
||||
from app.tasks.rotate_pdf_pages import rotate_pdf_pages
|
||||
from app.utils import log_task_progress
|
||||
from app.utils.ocr_provider import OCRResult, embed_text_layer, get_ocr_providers, merge_ocr_results
|
||||
from app.utils.text_quality import compare_text_quality
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@celery.task(base=BaseTaskWithRetry, bind=True)
|
||||
def process_with_ocr(self, filename: str, file_id: Optional[int] = None):
|
||||
def process_with_ocr(self, filename: str, file_id: Optional[int] = None, original_text: Optional[str] = None):
|
||||
"""Run the configured OCR providers on *filename* and continue the pipeline.
|
||||
|
||||
When multiple OCR providers are configured the results are merged using the
|
||||
AI model (or a simpler strategy controlled by ``OCR_MERGE_STRATEGY``).
|
||||
|
||||
If *original_text* is provided (the original embedded text that failed the
|
||||
quality check), the OCR result is compared against it using a head-to-head
|
||||
AI review. The higher-quality text is passed to downstream tasks.
|
||||
|
||||
Args:
|
||||
filename: Base name of the file inside ``<workdir>/tmp/``.
|
||||
file_id: Optional database record ID passed through to downstream tasks.
|
||||
original_text: Optional original embedded text for head-to-head comparison.
|
||||
"""
|
||||
task_id = self.request.id
|
||||
log_task_progress(
|
||||
@@ -138,22 +147,107 @@ def process_with_ocr(self, filename: str, file_id: Optional[int] = None):
|
||||
file_id=file_id,
|
||||
)
|
||||
|
||||
# ----------------------------------------------------------------
|
||||
# Head-to-head comparison with original embedded text (if provided)
|
||||
# ----------------------------------------------------------------
|
||||
final_text = extracted_text
|
||||
if original_text and original_text.strip() and extracted_text.strip():
|
||||
logger.info(f"[{task_id}] Original embedded text provided; running head-to-head quality comparison")
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"compare_ocr_quality",
|
||||
"in_progress",
|
||||
"Comparing OCR result against original embedded text",
|
||||
file_id=file_id,
|
||||
)
|
||||
try:
|
||||
comparison = compare_text_quality(original_text, extracted_text)
|
||||
comparison_detail = (
|
||||
f"Original score: {comparison.original_score}/100, "
|
||||
f"OCR score: {comparison.ocr_score}/100, "
|
||||
f"Preferred: {comparison.preferred}\n"
|
||||
f"AI explanation: {comparison.explanation}"
|
||||
)
|
||||
logger.info(f"[{task_id}] OCR comparison – {comparison_detail}")
|
||||
|
||||
if comparison.preferred == "original":
|
||||
# Original text is actually better – use it instead of OCR.
|
||||
final_text = original_text
|
||||
logger.info(
|
||||
f"[{task_id}] Original embedded text selected "
|
||||
f"(original={comparison.original_score} > ocr={comparison.ocr_score})"
|
||||
)
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"compare_ocr_quality",
|
||||
"success",
|
||||
f"Original text preferred (original={comparison.original_score}/100 vs "
|
||||
f"ocr={comparison.ocr_score}/100)",
|
||||
file_id=file_id,
|
||||
detail=comparison_detail,
|
||||
)
|
||||
else:
|
||||
logger.info(
|
||||
f"[{task_id}] OCR text selected "
|
||||
f"(preferred={comparison.preferred!r}, "
|
||||
f"ocr={comparison.ocr_score}, original={comparison.original_score})"
|
||||
)
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"compare_ocr_quality",
|
||||
"success",
|
||||
f"OCR text preferred (ocr={comparison.ocr_score}/100 vs "
|
||||
f"original={comparison.original_score}/100)",
|
||||
file_id=file_id,
|
||||
detail=comparison_detail,
|
||||
)
|
||||
except Exception as cmp_exc:
|
||||
logger.warning(f"[{task_id}] Head-to-head comparison failed ({cmp_exc}); keeping OCR text")
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"compare_ocr_quality",
|
||||
"skipped",
|
||||
f"Comparison failed ({cmp_exc}); keeping OCR output",
|
||||
file_id=file_id,
|
||||
)
|
||||
elif original_text is not None:
|
||||
# original_text was provided but one side is empty – pick whichever has content.
|
||||
if not extracted_text.strip() and original_text.strip():
|
||||
final_text = original_text
|
||||
logger.info(f"[{task_id}] OCR returned empty text; falling back to original embedded text")
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"compare_ocr_quality",
|
||||
"success",
|
||||
"OCR empty – using original embedded text",
|
||||
file_id=file_id,
|
||||
)
|
||||
else:
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"compare_ocr_quality",
|
||||
"skipped",
|
||||
"No original text to compare; using OCR output",
|
||||
file_id=file_id,
|
||||
)
|
||||
|
||||
log_task_progress(
|
||||
task_id,
|
||||
"process_with_ocr",
|
||||
"success",
|
||||
f"OCR complete for {filename}",
|
||||
file_id=file_id,
|
||||
detail=f"Extracted {len(extracted_text)} chars using {len(results)} provider(s)",
|
||||
detail=f"Extracted {len(extracted_text)} chars using {len(results)} provider(s); "
|
||||
f"final text length: {len(final_text)} chars",
|
||||
)
|
||||
|
||||
# Continue pipeline: rotate pages (if needed), then extract metadata
|
||||
rotate_pdf_pages.delay(filename, extracted_text, rotation_data, file_id)
|
||||
rotate_pdf_pages.delay(filename, final_text, rotation_data, file_id)
|
||||
|
||||
return {
|
||||
"file": filename,
|
||||
"searchable_pdf": searchable_pdf_path or tmp_file_path,
|
||||
"cleaned_text": extracted_text,
|
||||
"cleaned_text": final_text,
|
||||
"providers_used": [r.provider for r in results],
|
||||
}
|
||||
|
||||
|
||||
@@ -1070,6 +1070,31 @@ SETTING_METADATA = {
|
||||
"required": False,
|
||||
"restart_required": False,
|
||||
},
|
||||
"text_quality_threshold": {
|
||||
"category": "Processing",
|
||||
"description": (
|
||||
"Minimum quality score (0–100) required to accept embedded PDF text without re-OCR. "
|
||||
"Text scoring below this threshold triggers a fresh OCR pass. "
|
||||
"Default: 85. Lower values are more permissive; higher values enforce stricter quality."
|
||||
),
|
||||
"type": "integer",
|
||||
"sensitive": False,
|
||||
"required": False,
|
||||
"restart_required": False,
|
||||
},
|
||||
"text_quality_significant_issues": {
|
||||
"category": "Processing",
|
||||
"description": (
|
||||
"Comma-separated list of quality issue labels that force OCR re-run even when the quality "
|
||||
"score meets TEXT_QUALITY_THRESHOLD. Any matching issue in the AI assessment will trigger "
|
||||
"re-OCR regardless of the numeric score. "
|
||||
"Default: excessive_typos,garbage_characters,incoherent_text,fragmented_sentences"
|
||||
),
|
||||
"type": "list",
|
||||
"sensitive": False,
|
||||
"required": False,
|
||||
"restart_required": False,
|
||||
},
|
||||
# Notifications Settings
|
||||
"notification_urls": {
|
||||
"category": "Notifications",
|
||||
|
||||
+152
-9
@@ -5,6 +5,7 @@ This module provides functionality to:
|
||||
- Detect whether a PDF's embedded text came from a digital creation process
|
||||
(e.g., exported from Word, LibreOffice, LaTeX) or a previous OCR pass.
|
||||
- Assess the quality of extracted text using an AI model.
|
||||
- Compare two candidate text extractions and choose the higher-quality one.
|
||||
- Log detailed feedback for debugging and continuous improvement.
|
||||
|
||||
**Rationale**
|
||||
@@ -19,6 +20,10 @@ be good and no AI check is performed. If the text appears to come from a prior
|
||||
OCR pass (or the source is unknown), the AI quality check is performed. Poor
|
||||
quality text triggers automatic re-OCR so that the downstream pipeline operates
|
||||
on the best available text.
|
||||
|
||||
After re-OCR, the new text is scored and compared head-to-head against the
|
||||
original embedded text via :func:`compare_text_quality` to ensure the pipeline
|
||||
always uses the highest-quality extraction available.
|
||||
"""
|
||||
|
||||
import json
|
||||
@@ -107,7 +112,7 @@ class TextSource(str, Enum):
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Result data class
|
||||
# Result data classes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@@ -123,6 +128,17 @@ class TextQualityResult:
|
||||
ai_response_raw: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class TextComparisonResult:
|
||||
"""Result of a head-to-head comparison between two candidate texts."""
|
||||
|
||||
preferred: str # "original" | "ocr" | "equal"
|
||||
original_score: int
|
||||
ocr_score: int
|
||||
explanation: str
|
||||
ai_response_raw: Optional[str] = None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Public API
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -191,6 +207,15 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
|
||||
- Lack of semantic coherence.
|
||||
- Garbage characters or symbol soup.
|
||||
|
||||
The acceptance criteria are controlled by two settings:
|
||||
|
||||
- ``settings.text_quality_threshold`` – minimum score (default 85) for
|
||||
auto-acceptance.
|
||||
- ``settings.text_quality_significant_issues`` – list of issue labels that
|
||||
force re-OCR even when the score meets the threshold (e.g.
|
||||
``excessive_typos``, ``garbage_characters``, ``incoherent_text``,
|
||||
``fragmented_sentences``).
|
||||
|
||||
The text sample and the full AI feedback are logged at DEBUG / INFO level
|
||||
to aid debugging and continuous quality improvement.
|
||||
|
||||
@@ -226,10 +251,21 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
|
||||
issues=["empty_text"],
|
||||
)
|
||||
|
||||
# 3. Retrieve configurable thresholds.
|
||||
threshold = getattr(settings, "text_quality_threshold", 85)
|
||||
significant_issues: list[str] = list(
|
||||
getattr(
|
||||
settings,
|
||||
"text_quality_significant_issues",
|
||||
["excessive_typos", "garbage_characters", "incoherent_text", "fragmented_sentences"],
|
||||
)
|
||||
)
|
||||
|
||||
sample = stripped[:_TEXT_SAMPLE_MAX_CHARS]
|
||||
logger.info(
|
||||
f"[text_quality] Assessing text quality "
|
||||
f"(source={text_source.value}, sample_chars={len(sample)}, total_chars={len(stripped)})"
|
||||
f"(source={text_source.value}, sample_chars={len(sample)}, total_chars={len(stripped)}, "
|
||||
f"threshold={threshold})"
|
||||
)
|
||||
logger.debug(f"[text_quality] Text sample forwarded to AI:\n{sample}")
|
||||
|
||||
@@ -240,20 +276,21 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
|
||||
"nonsensical fragments.\n\n"
|
||||
"Evaluate the following text and return a JSON object with exactly these fields:\n"
|
||||
' "quality_score": integer 0-100 (0=completely garbled, 100=perfect text)\n'
|
||||
' "is_good_quality": boolean (true if quality_score >= 65)\n'
|
||||
f' "is_good_quality": boolean (true if quality_score >= {threshold} AND no significant issues)\n'
|
||||
' "feedback": one-sentence summary of your assessment\n'
|
||||
' "issues": list of issues found (e.g. ["excessive_typos", "garbage_characters", '
|
||||
'"incoherent_text", "fragmented_sentences"]); empty list if none\n\n'
|
||||
"Criteria for POOR quality (score < 65):\n"
|
||||
f"Criteria for POOR quality (score < {threshold}):\n"
|
||||
"- Excessive typos, misspellings, or letter substitutions typical of OCR errors\n"
|
||||
"- Garbage characters (%, @, #, symbols mixed randomly into words)\n"
|
||||
"- Incoherent or nonsensical sentences that carry no meaning\n"
|
||||
"- Sequences of random characters or numbers without context\n"
|
||||
"- Heavy fragmentation (isolated letters or words without sentence structure)\n\n"
|
||||
"Criteria for GOOD quality (score >= 65):\n"
|
||||
"- Mostly readable text with at most minor imperfections\n"
|
||||
f"Criteria for GOOD quality (score >= {threshold}):\n"
|
||||
"- Mostly readable text with at most very minor imperfections\n"
|
||||
"- Coherent sentences and/or paragraphs\n"
|
||||
"- Recognisable language (any language accepted)\n\n"
|
||||
"- Recognisable language (any language accepted)\n"
|
||||
"- No significant OCR artefacts\n\n"
|
||||
f"Text to evaluate:\n---\n{sample}\n---\n\n"
|
||||
"Return only the JSON object, no markdown fences."
|
||||
)
|
||||
@@ -281,13 +318,28 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
|
||||
parsed: dict = json.loads(clean)
|
||||
|
||||
quality_score = int(parsed.get("quality_score", 0))
|
||||
is_good = bool(parsed.get("is_good_quality", quality_score >= 65))
|
||||
is_good_ai = bool(parsed.get("is_good_quality", quality_score >= threshold))
|
||||
feedback = str(parsed.get("feedback", ""))
|
||||
issues = list(parsed.get("issues", []))
|
||||
|
||||
# Apply strict rules: reject when score is below threshold OR when any
|
||||
# significant issue is present (even if the AI says is_good_quality=true).
|
||||
score_ok = quality_score >= threshold
|
||||
has_significant_issue = bool(significant_issues and any(i in issues for i in significant_issues))
|
||||
|
||||
if has_significant_issue and is_good_ai:
|
||||
logger.warning(
|
||||
f"[text_quality] Overriding AI is_good_quality=True – significant issues present: "
|
||||
f"{[i for i in issues if i in significant_issues]} (score={quality_score})"
|
||||
)
|
||||
|
||||
is_good = score_ok and is_good_ai and not has_significant_issue
|
||||
|
||||
logger.info(
|
||||
f"[text_quality] Quality assessment complete – "
|
||||
f"score={quality_score}, good={is_good}, issues={issues}, feedback={feedback!r}"
|
||||
f"score={quality_score}, threshold={threshold}, score_ok={score_ok}, "
|
||||
f"ai_good={is_good_ai}, significant_issues_found={has_significant_issue}, "
|
||||
f"final_good={is_good}, issues={issues}, feedback={feedback!r}"
|
||||
)
|
||||
|
||||
return TextQualityResult(
|
||||
@@ -324,3 +376,94 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
|
||||
feedback=f"Quality check could not be performed ({exc}); assuming acceptable quality.",
|
||||
ai_response_raw=response_text,
|
||||
)
|
||||
|
||||
|
||||
def compare_text_quality(original_text: str, ocr_text: str) -> TextComparisonResult:
|
||||
"""Compare the quality of two candidate text extractions side-by-side using AI.
|
||||
|
||||
Used after a re-OCR pass to decide whether the new OCR output is actually
|
||||
better than the original embedded text. The AI evaluates both texts
|
||||
independently and then picks the preferred one.
|
||||
|
||||
Args:
|
||||
original_text: Text extracted from the PDF's original embedded layer.
|
||||
ocr_text: Text produced by the re-OCR pipeline.
|
||||
|
||||
Returns:
|
||||
:class:`TextComparisonResult` indicating which text is preferred and why.
|
||||
"""
|
||||
orig_sample = original_text.strip()[:_TEXT_SAMPLE_MAX_CHARS]
|
||||
ocr_sample = ocr_text.strip()[:_TEXT_SAMPLE_MAX_CHARS]
|
||||
|
||||
logger.info(f"[text_quality] Comparing original ({len(orig_sample)} chars) vs OCR ({len(ocr_sample)} chars) texts")
|
||||
|
||||
prompt = (
|
||||
"You are a document quality assessor comparing two text extractions from the same PDF.\n\n"
|
||||
"TEXT A (original embedded text):\n"
|
||||
f"---\n{orig_sample}\n---\n\n"
|
||||
"TEXT B (re-OCR text):\n"
|
||||
f"---\n{ocr_sample}\n---\n\n"
|
||||
"Score each text independently (0–100) and decide which is better for downstream "
|
||||
"document processing (metadata extraction, search, AI analysis).\n\n"
|
||||
"Return a JSON object with exactly these fields:\n"
|
||||
' "original_score": integer 0-100 for TEXT A\n'
|
||||
' "ocr_score": integer 0-100 for TEXT B\n'
|
||||
' "preferred": one of "original", "ocr", or "equal"\n'
|
||||
' "explanation": one-sentence rationale\n\n'
|
||||
"Return only the JSON object, no markdown fences."
|
||||
)
|
||||
|
||||
response_text: Optional[str] = None
|
||||
try:
|
||||
provider = get_ai_provider()
|
||||
model = settings.ai_model or settings.openai_model or "gpt-4o-mini"
|
||||
response_text = provider.chat_completion(
|
||||
messages=[
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You are a document quality assessor. Respond only with valid JSON.",
|
||||
},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
model=model,
|
||||
temperature=0,
|
||||
)
|
||||
|
||||
logger.info(f"[text_quality] AI comparison raw response: {response_text[:500]}")
|
||||
|
||||
clean = re.sub(r"```(?:json)?\s*", "", response_text).strip().rstrip("`").strip()
|
||||
parsed: dict = json.loads(clean)
|
||||
|
||||
original_score = int(parsed.get("original_score", 0))
|
||||
ocr_score = int(parsed.get("ocr_score", 0))
|
||||
preferred = str(parsed.get("preferred", "ocr"))
|
||||
explanation = str(parsed.get("explanation", ""))
|
||||
|
||||
if preferred not in ("original", "ocr", "equal"):
|
||||
logger.warning(f"[text_quality] Unexpected preferred value {preferred!r}; defaulting to 'ocr'")
|
||||
preferred = "ocr"
|
||||
|
||||
logger.info(
|
||||
f"[text_quality] Comparison result – original={original_score}, ocr={ocr_score}, "
|
||||
f"preferred={preferred!r}, explanation={explanation!r}"
|
||||
)
|
||||
|
||||
return TextComparisonResult(
|
||||
preferred=preferred,
|
||||
original_score=original_score,
|
||||
ocr_score=ocr_score,
|
||||
explanation=explanation,
|
||||
ai_response_raw=response_text,
|
||||
)
|
||||
|
||||
except Exception as exc:
|
||||
logger.warning(f"[text_quality] Comparison failed ({exc}); defaulting to OCR text.")
|
||||
# Safe fallback: if comparison fails, keep the OCR result (which was
|
||||
# triggered because the original text was already deemed poor).
|
||||
return TextComparisonResult(
|
||||
preferred="ocr",
|
||||
original_score=0,
|
||||
ocr_score=0,
|
||||
explanation=f"Comparison could not be performed ({exc}); defaulting to OCR output.",
|
||||
ai_response_raw=response_text,
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user