Merge pull request #405 from christianlouis/copilot/fine-tune-ocr-criteria

feat(ocr): Fine-tune OCR quality acceptance criteria with strict threshold and head-to-head comparison
This commit is contained in:
Christian Krakau-Louis
2026-02-25 14:17:43 +01:00
committed by GitHub
7 changed files with 659 additions and 46 deletions
+28
View File
@@ -288,6 +288,22 @@ class Settings(BaseSettings):
"bypass the check. Default: True (enabled)." "bypass the check. Default: True (enabled)."
), ),
) )
text_quality_threshold: int = Field(
default=85,
description=(
"Minimum quality score (0100) required to accept embedded PDF text without re-OCR. "
"Text scoring below this threshold is discarded and the file is re-processed with OCR. "
"Default: 85. The stricter this value, the more files will be re-OCR'd."
),
)
text_quality_significant_issues: Union[List[str], str] = Field(
default_factory=lambda: ["excessive_typos", "garbage_characters", "incoherent_text", "fragmented_sentences"],
description=(
"Comma-separated list of quality issue labels that force OCR re-run even when the quality "
"score is above TEXT_QUALITY_THRESHOLD. Any of these issues present in the AI assessment "
"will trigger re-OCR. Default: excessive_typos,garbage_characters,incoherent_text,fragmented_sentences"
),
)
# Processing step timeout - prevents files from getting stuck in "in_progress" state # Processing step timeout - prevents files from getting stuck in "in_progress" state
step_timeout: int = Field( step_timeout: int = Field(
@@ -443,6 +459,18 @@ class Settings(BaseSettings):
return [] return []
return v return v
@field_validator("text_quality_significant_issues", mode="before")
@classmethod
def parse_text_quality_significant_issues(cls, v: str | list[str]) -> list[str]:
"""Parse significant issue labels from comma-separated string or list."""
if isinstance(v, str):
if "," in v:
return [item.strip() for item in v.split(",") if item.strip()]
elif v.strip():
return [v.strip()]
return []
return v
@field_validator("cors_allowed_origins", "cors_allowed_methods", "cors_allowed_headers", mode="before") @field_validator("cors_allowed_origins", "cors_allowed_methods", "cors_allowed_headers", mode="before")
@classmethod @classmethod
def parse_comma_separated_list(cls, v: str | list[str]) -> list[str]: def parse_comma_separated_list(cls, v: str | list[str]) -> list[str]:
+5 -3
View File
@@ -428,19 +428,21 @@ def process_document(
if not quality_result.is_good_quality: if not quality_result.is_good_quality:
# Poor quality: discard embedded text and re-OCR instead. # Poor quality: discard embedded text and re-OCR instead.
# Pass the original embedded text so the OCR task can compare
# its result against the original and keep the better version.
issues_str = ", ".join(quality_result.issues) if quality_result.issues else "unspecified" issues_str = ", ".join(quality_result.issues) if quality_result.issues else "unspecified"
detail_msg = ( detail_msg = (
f"Text quality check FAILED score={quality_result.quality_score}/100, " f"Text quality check FAILED score={quality_result.quality_score}/100, "
f"source={quality_result.text_source.value}, issues=[{issues_str}].\n" f"source={quality_result.text_source.value}, issues=[{issues_str}].\n"
f"AI feedback: {quality_result.feedback}\n" f"AI feedback: {quality_result.feedback}\n"
f"Embedded text will be ignored; re-running OCR." f"Embedded text will be compared with fresh OCR output; best version will be used."
) )
logger.warning(f"[{task_id}] {detail_msg}") logger.warning(f"[{task_id}] {detail_msg}")
log_task_progress( log_task_progress(
task_id, task_id,
"check_text_quality", "check_text_quality",
"failure", "failure",
f"Poor quality text (score={quality_result.quality_score}/100); queuing OCR", f"Poor quality text (score={quality_result.quality_score}/100); queuing OCR for comparison",
file_id=file_id, file_id=file_id,
detail=detail_msg, detail=detail_msg,
) )
@@ -451,7 +453,7 @@ def process_document(
"Queued for OCR (text quality too low)", "Queued for OCR (text quality too low)",
file_id=file_id, file_id=file_id,
) )
process_with_ocr.delay(new_filename, file_id) process_with_ocr.delay(new_filename, file_id, extracted_text)
return { return {
"file": new_local_path, "file": new_local_path,
"status": "Queued for OCR (poor embedded text quality)", "status": "Queued for OCR (poor embedded text quality)",
+99 -5
View File
@@ -8,7 +8,10 @@ task with a multi-engine OCR pipeline that:
2. Merges/cross-checks the results using the configured AI model when more 2. Merges/cross-checks the results using the configured AI model when more
than one provider is active (see ``OCR_MERGE_STRATEGY``). than one provider is active (see ``OCR_MERGE_STRATEGY``).
3. Writes the best searchable PDF back to the working directory. 3. Writes the best searchable PDF back to the working directory.
4. Hands off to the page-rotation and metadata-extraction pipeline exactly as 4. Optionally compares the OCR output against the original embedded text
(passed as *original_text*) using a head-to-head AI review and keeps the
higher-quality text for downstream processing.
5. Hands off to the page-rotation and metadata-extraction pipeline exactly as
the legacy Azure task did. the legacy Azure task did.
""" """
@@ -22,20 +25,26 @@ from app.tasks.retry_config import BaseTaskWithRetry
from app.tasks.rotate_pdf_pages import rotate_pdf_pages from app.tasks.rotate_pdf_pages import rotate_pdf_pages
from app.utils import log_task_progress from app.utils import log_task_progress
from app.utils.ocr_provider import OCRResult, embed_text_layer, get_ocr_providers, merge_ocr_results from app.utils.ocr_provider import OCRResult, embed_text_layer, get_ocr_providers, merge_ocr_results
from app.utils.text_quality import compare_text_quality
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@celery.task(base=BaseTaskWithRetry, bind=True) @celery.task(base=BaseTaskWithRetry, bind=True)
def process_with_ocr(self, filename: str, file_id: Optional[int] = None): def process_with_ocr(self, filename: str, file_id: Optional[int] = None, original_text: Optional[str] = None):
"""Run the configured OCR providers on *filename* and continue the pipeline. """Run the configured OCR providers on *filename* and continue the pipeline.
When multiple OCR providers are configured the results are merged using the When multiple OCR providers are configured the results are merged using the
AI model (or a simpler strategy controlled by ``OCR_MERGE_STRATEGY``). AI model (or a simpler strategy controlled by ``OCR_MERGE_STRATEGY``).
If *original_text* is provided (the original embedded text that failed the
quality check), the OCR result is compared against it using a head-to-head
AI review. The higher-quality text is passed to downstream tasks.
Args: Args:
filename: Base name of the file inside ``<workdir>/tmp/``. filename: Base name of the file inside ``<workdir>/tmp/``.
file_id: Optional database record ID passed through to downstream tasks. file_id: Optional database record ID passed through to downstream tasks.
original_text: Optional original embedded text for head-to-head comparison.
""" """
task_id = self.request.id task_id = self.request.id
log_task_progress( log_task_progress(
@@ -138,22 +147,107 @@ def process_with_ocr(self, filename: str, file_id: Optional[int] = None):
file_id=file_id, file_id=file_id,
) )
# ----------------------------------------------------------------
# Head-to-head comparison with original embedded text (if provided)
# ----------------------------------------------------------------
final_text = extracted_text
if original_text and original_text.strip() and extracted_text.strip():
logger.info(f"[{task_id}] Original embedded text provided; running head-to-head quality comparison")
log_task_progress(
task_id,
"compare_ocr_quality",
"in_progress",
"Comparing OCR result against original embedded text",
file_id=file_id,
)
try:
comparison = compare_text_quality(original_text, extracted_text)
comparison_detail = (
f"Original score: {comparison.original_score}/100, "
f"OCR score: {comparison.ocr_score}/100, "
f"Preferred: {comparison.preferred}\n"
f"AI explanation: {comparison.explanation}"
)
logger.info(f"[{task_id}] OCR comparison {comparison_detail}")
if comparison.preferred == "original":
# Original text is actually better use it instead of OCR.
final_text = original_text
logger.info(
f"[{task_id}] Original embedded text selected "
f"(original={comparison.original_score} > ocr={comparison.ocr_score})"
)
log_task_progress(
task_id,
"compare_ocr_quality",
"success",
f"Original text preferred (original={comparison.original_score}/100 vs "
f"ocr={comparison.ocr_score}/100)",
file_id=file_id,
detail=comparison_detail,
)
else:
logger.info(
f"[{task_id}] OCR text selected "
f"(preferred={comparison.preferred!r}, "
f"ocr={comparison.ocr_score}, original={comparison.original_score})"
)
log_task_progress(
task_id,
"compare_ocr_quality",
"success",
f"OCR text preferred (ocr={comparison.ocr_score}/100 vs "
f"original={comparison.original_score}/100)",
file_id=file_id,
detail=comparison_detail,
)
except Exception as cmp_exc:
logger.warning(f"[{task_id}] Head-to-head comparison failed ({cmp_exc}); keeping OCR text")
log_task_progress(
task_id,
"compare_ocr_quality",
"skipped",
f"Comparison failed ({cmp_exc}); keeping OCR output",
file_id=file_id,
)
elif original_text is not None:
# original_text was provided but one side is empty pick whichever has content.
if not extracted_text.strip() and original_text.strip():
final_text = original_text
logger.info(f"[{task_id}] OCR returned empty text; falling back to original embedded text")
log_task_progress(
task_id,
"compare_ocr_quality",
"success",
"OCR empty using original embedded text",
file_id=file_id,
)
else:
log_task_progress(
task_id,
"compare_ocr_quality",
"skipped",
"No original text to compare; using OCR output",
file_id=file_id,
)
log_task_progress( log_task_progress(
task_id, task_id,
"process_with_ocr", "process_with_ocr",
"success", "success",
f"OCR complete for {filename}", f"OCR complete for {filename}",
file_id=file_id, file_id=file_id,
detail=f"Extracted {len(extracted_text)} chars using {len(results)} provider(s)", detail=f"Extracted {len(extracted_text)} chars using {len(results)} provider(s); "
f"final text length: {len(final_text)} chars",
) )
# Continue pipeline: rotate pages (if needed), then extract metadata # Continue pipeline: rotate pages (if needed), then extract metadata
rotate_pdf_pages.delay(filename, extracted_text, rotation_data, file_id) rotate_pdf_pages.delay(filename, final_text, rotation_data, file_id)
return { return {
"file": filename, "file": filename,
"searchable_pdf": searchable_pdf_path or tmp_file_path, "searchable_pdf": searchable_pdf_path or tmp_file_path,
"cleaned_text": extracted_text, "cleaned_text": final_text,
"providers_used": [r.provider for r in results], "providers_used": [r.provider for r in results],
} }
+25
View File
@@ -1070,6 +1070,31 @@ SETTING_METADATA = {
"required": False, "required": False,
"restart_required": False, "restart_required": False,
}, },
"text_quality_threshold": {
"category": "Processing",
"description": (
"Minimum quality score (0100) required to accept embedded PDF text without re-OCR. "
"Text scoring below this threshold triggers a fresh OCR pass. "
"Default: 85. Lower values are more permissive; higher values enforce stricter quality."
),
"type": "integer",
"sensitive": False,
"required": False,
"restart_required": False,
},
"text_quality_significant_issues": {
"category": "Processing",
"description": (
"Comma-separated list of quality issue labels that force OCR re-run even when the quality "
"score meets TEXT_QUALITY_THRESHOLD. Any matching issue in the AI assessment will trigger "
"re-OCR regardless of the numeric score. "
"Default: excessive_typos,garbage_characters,incoherent_text,fragmented_sentences"
),
"type": "list",
"sensitive": False,
"required": False,
"restart_required": False,
},
# Notifications Settings # Notifications Settings
"notification_urls": { "notification_urls": {
"category": "Notifications", "category": "Notifications",
+152 -9
View File
@@ -5,6 +5,7 @@ This module provides functionality to:
- Detect whether a PDF's embedded text came from a digital creation process - Detect whether a PDF's embedded text came from a digital creation process
(e.g., exported from Word, LibreOffice, LaTeX) or a previous OCR pass. (e.g., exported from Word, LibreOffice, LaTeX) or a previous OCR pass.
- Assess the quality of extracted text using an AI model. - Assess the quality of extracted text using an AI model.
- Compare two candidate text extractions and choose the higher-quality one.
- Log detailed feedback for debugging and continuous improvement. - Log detailed feedback for debugging and continuous improvement.
**Rationale** **Rationale**
@@ -19,6 +20,10 @@ be good and no AI check is performed. If the text appears to come from a prior
OCR pass (or the source is unknown), the AI quality check is performed. Poor OCR pass (or the source is unknown), the AI quality check is performed. Poor
quality text triggers automatic re-OCR so that the downstream pipeline operates quality text triggers automatic re-OCR so that the downstream pipeline operates
on the best available text. on the best available text.
After re-OCR, the new text is scored and compared head-to-head against the
original embedded text via :func:`compare_text_quality` to ensure the pipeline
always uses the highest-quality extraction available.
""" """
import json import json
@@ -107,7 +112,7 @@ class TextSource(str, Enum):
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Result data class # Result data classes
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -123,6 +128,17 @@ class TextQualityResult:
ai_response_raw: Optional[str] = None ai_response_raw: Optional[str] = None
@dataclass
class TextComparisonResult:
"""Result of a head-to-head comparison between two candidate texts."""
preferred: str # "original" | "ocr" | "equal"
original_score: int
ocr_score: int
explanation: str
ai_response_raw: Optional[str] = None
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Public API # Public API
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -191,6 +207,15 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
- Lack of semantic coherence. - Lack of semantic coherence.
- Garbage characters or symbol soup. - Garbage characters or symbol soup.
The acceptance criteria are controlled by two settings:
- ``settings.text_quality_threshold`` minimum score (default 85) for
auto-acceptance.
- ``settings.text_quality_significant_issues`` list of issue labels that
force re-OCR even when the score meets the threshold (e.g.
``excessive_typos``, ``garbage_characters``, ``incoherent_text``,
``fragmented_sentences``).
The text sample and the full AI feedback are logged at DEBUG / INFO level The text sample and the full AI feedback are logged at DEBUG / INFO level
to aid debugging and continuous quality improvement. to aid debugging and continuous quality improvement.
@@ -226,10 +251,21 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
issues=["empty_text"], issues=["empty_text"],
) )
# 3. Retrieve configurable thresholds.
threshold = getattr(settings, "text_quality_threshold", 85)
significant_issues: list[str] = list(
getattr(
settings,
"text_quality_significant_issues",
["excessive_typos", "garbage_characters", "incoherent_text", "fragmented_sentences"],
)
)
sample = stripped[:_TEXT_SAMPLE_MAX_CHARS] sample = stripped[:_TEXT_SAMPLE_MAX_CHARS]
logger.info( logger.info(
f"[text_quality] Assessing text quality " f"[text_quality] Assessing text quality "
f"(source={text_source.value}, sample_chars={len(sample)}, total_chars={len(stripped)})" f"(source={text_source.value}, sample_chars={len(sample)}, total_chars={len(stripped)}, "
f"threshold={threshold})"
) )
logger.debug(f"[text_quality] Text sample forwarded to AI:\n{sample}") logger.debug(f"[text_quality] Text sample forwarded to AI:\n{sample}")
@@ -240,20 +276,21 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
"nonsensical fragments.\n\n" "nonsensical fragments.\n\n"
"Evaluate the following text and return a JSON object with exactly these fields:\n" "Evaluate the following text and return a JSON object with exactly these fields:\n"
' "quality_score": integer 0-100 (0=completely garbled, 100=perfect text)\n' ' "quality_score": integer 0-100 (0=completely garbled, 100=perfect text)\n'
' "is_good_quality": boolean (true if quality_score >= 65)\n' f' "is_good_quality": boolean (true if quality_score >= {threshold} AND no significant issues)\n'
' "feedback": one-sentence summary of your assessment\n' ' "feedback": one-sentence summary of your assessment\n'
' "issues": list of issues found (e.g. ["excessive_typos", "garbage_characters", ' ' "issues": list of issues found (e.g. ["excessive_typos", "garbage_characters", '
'"incoherent_text", "fragmented_sentences"]); empty list if none\n\n' '"incoherent_text", "fragmented_sentences"]); empty list if none\n\n'
"Criteria for POOR quality (score < 65):\n" f"Criteria for POOR quality (score < {threshold}):\n"
"- Excessive typos, misspellings, or letter substitutions typical of OCR errors\n" "- Excessive typos, misspellings, or letter substitutions typical of OCR errors\n"
"- Garbage characters (%, @, #, symbols mixed randomly into words)\n" "- Garbage characters (%, @, #, symbols mixed randomly into words)\n"
"- Incoherent or nonsensical sentences that carry no meaning\n" "- Incoherent or nonsensical sentences that carry no meaning\n"
"- Sequences of random characters or numbers without context\n" "- Sequences of random characters or numbers without context\n"
"- Heavy fragmentation (isolated letters or words without sentence structure)\n\n" "- Heavy fragmentation (isolated letters or words without sentence structure)\n\n"
"Criteria for GOOD quality (score >= 65):\n" f"Criteria for GOOD quality (score >= {threshold}):\n"
"- Mostly readable text with at most minor imperfections\n" "- Mostly readable text with at most very minor imperfections\n"
"- Coherent sentences and/or paragraphs\n" "- Coherent sentences and/or paragraphs\n"
"- Recognisable language (any language accepted)\n\n" "- Recognisable language (any language accepted)\n"
"- No significant OCR artefacts\n\n"
f"Text to evaluate:\n---\n{sample}\n---\n\n" f"Text to evaluate:\n---\n{sample}\n---\n\n"
"Return only the JSON object, no markdown fences." "Return only the JSON object, no markdown fences."
) )
@@ -281,13 +318,28 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
parsed: dict = json.loads(clean) parsed: dict = json.loads(clean)
quality_score = int(parsed.get("quality_score", 0)) quality_score = int(parsed.get("quality_score", 0))
is_good = bool(parsed.get("is_good_quality", quality_score >= 65)) is_good_ai = bool(parsed.get("is_good_quality", quality_score >= threshold))
feedback = str(parsed.get("feedback", "")) feedback = str(parsed.get("feedback", ""))
issues = list(parsed.get("issues", [])) issues = list(parsed.get("issues", []))
# Apply strict rules: reject when score is below threshold OR when any
# significant issue is present (even if the AI says is_good_quality=true).
score_ok = quality_score >= threshold
has_significant_issue = bool(significant_issues and any(i in issues for i in significant_issues))
if has_significant_issue and is_good_ai:
logger.warning(
f"[text_quality] Overriding AI is_good_quality=True significant issues present: "
f"{[i for i in issues if i in significant_issues]} (score={quality_score})"
)
is_good = score_ok and is_good_ai and not has_significant_issue
logger.info( logger.info(
f"[text_quality] Quality assessment complete " f"[text_quality] Quality assessment complete "
f"score={quality_score}, good={is_good}, issues={issues}, feedback={feedback!r}" f"score={quality_score}, threshold={threshold}, score_ok={score_ok}, "
f"ai_good={is_good_ai}, significant_issues_found={has_significant_issue}, "
f"final_good={is_good}, issues={issues}, feedback={feedback!r}"
) )
return TextQualityResult( return TextQualityResult(
@@ -324,3 +376,94 @@ def check_text_quality(text: str, text_source: TextSource) -> TextQualityResult:
feedback=f"Quality check could not be performed ({exc}); assuming acceptable quality.", feedback=f"Quality check could not be performed ({exc}); assuming acceptable quality.",
ai_response_raw=response_text, ai_response_raw=response_text,
) )
def compare_text_quality(original_text: str, ocr_text: str) -> TextComparisonResult:
"""Compare the quality of two candidate text extractions side-by-side using AI.
Used after a re-OCR pass to decide whether the new OCR output is actually
better than the original embedded text. The AI evaluates both texts
independently and then picks the preferred one.
Args:
original_text: Text extracted from the PDF's original embedded layer.
ocr_text: Text produced by the re-OCR pipeline.
Returns:
:class:`TextComparisonResult` indicating which text is preferred and why.
"""
orig_sample = original_text.strip()[:_TEXT_SAMPLE_MAX_CHARS]
ocr_sample = ocr_text.strip()[:_TEXT_SAMPLE_MAX_CHARS]
logger.info(f"[text_quality] Comparing original ({len(orig_sample)} chars) vs OCR ({len(ocr_sample)} chars) texts")
prompt = (
"You are a document quality assessor comparing two text extractions from the same PDF.\n\n"
"TEXT A (original embedded text):\n"
f"---\n{orig_sample}\n---\n\n"
"TEXT B (re-OCR text):\n"
f"---\n{ocr_sample}\n---\n\n"
"Score each text independently (0100) and decide which is better for downstream "
"document processing (metadata extraction, search, AI analysis).\n\n"
"Return a JSON object with exactly these fields:\n"
' "original_score": integer 0-100 for TEXT A\n'
' "ocr_score": integer 0-100 for TEXT B\n'
' "preferred": one of "original", "ocr", or "equal"\n'
' "explanation": one-sentence rationale\n\n'
"Return only the JSON object, no markdown fences."
)
response_text: Optional[str] = None
try:
provider = get_ai_provider()
model = settings.ai_model or settings.openai_model or "gpt-4o-mini"
response_text = provider.chat_completion(
messages=[
{
"role": "system",
"content": "You are a document quality assessor. Respond only with valid JSON.",
},
{"role": "user", "content": prompt},
],
model=model,
temperature=0,
)
logger.info(f"[text_quality] AI comparison raw response: {response_text[:500]}")
clean = re.sub(r"```(?:json)?\s*", "", response_text).strip().rstrip("`").strip()
parsed: dict = json.loads(clean)
original_score = int(parsed.get("original_score", 0))
ocr_score = int(parsed.get("ocr_score", 0))
preferred = str(parsed.get("preferred", "ocr"))
explanation = str(parsed.get("explanation", ""))
if preferred not in ("original", "ocr", "equal"):
logger.warning(f"[text_quality] Unexpected preferred value {preferred!r}; defaulting to 'ocr'")
preferred = "ocr"
logger.info(
f"[text_quality] Comparison result original={original_score}, ocr={ocr_score}, "
f"preferred={preferred!r}, explanation={explanation!r}"
)
return TextComparisonResult(
preferred=preferred,
original_score=original_score,
ocr_score=ocr_score,
explanation=explanation,
ai_response_raw=response_text,
)
except Exception as exc:
logger.warning(f"[text_quality] Comparison failed ({exc}); defaulting to OCR text.")
# Safe fallback: if comparison fails, keep the OCR result (which was
# triggered because the original text was already deemed poor).
return TextComparisonResult(
preferred="ocr",
original_score=0,
ocr_score=0,
explanation=f"Comparison could not be performed ({exc}); defaulting to OCR output.",
ai_response_raw=response_text,
)
+12 -5
View File
@@ -465,9 +465,11 @@ When multiple providers are listed, all run in parallel and their results are me
DocuElevate can automatically assess whether the text already embedded in a PDF is of sufficient quality before deciding to skip OCR. This prevents poor OCR output from a previous scan being silently used for downstream processing. DocuElevate can automatically assess whether the text already embedded in a PDF is of sufficient quality before deciding to skip OCR. This prevents poor OCR output from a previous scan being silently used for downstream processing.
| **Variable** | **Description** | **Default** | | **Variable** | **Description** | **Default** |
|--------------------------------|---------------------------------------------------------------------------------|-------------| |-------------------------------------------|---------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------------|
| `ENABLE_TEXT_QUALITY_CHECK` | Enable AI-based quality assessment of embedded PDF text. | `true` | | `ENABLE_TEXT_QUALITY_CHECK` | Enable AI-based quality assessment of embedded PDF text. | `true` |
| `TEXT_QUALITY_THRESHOLD` | Minimum quality score (0100) required to accept embedded text without re-OCR. | `85` |
| `TEXT_QUALITY_SIGNIFICANT_ISSUES` | Comma-separated issue labels that force re-OCR even when the score meets the threshold. | `excessive_typos,garbage_characters,incoherent_text,fragmented_sentences` |
**How it works:** **How it works:**
@@ -478,11 +480,16 @@ DocuElevate can automatically assess whether the text already embedded in a PDF
- Garbage characters or symbol soup - Garbage characters or symbol soup
- Incoherent or nonsensical sentences - Incoherent or nonsensical sentences
- Heavy fragmentation - Heavy fragmentation
4. If the quality score falls **below 65/100**, the embedded text is discarded and the file is sent to the configured OCR providers for a fresh scan. 4. The text is **rejected** (and re-OCR triggered) when **either** of these conditions is true:
5. All quality decisions (score, source, AI feedback) are recorded in the processing log for review. - The quality score is **below** `TEXT_QUALITY_THRESHOLD` (default 85), **or**
- The AI identifies one or more issues listed in `TEXT_QUALITY_SIGNIFICANT_ISSUES` — even if the numeric score is above the threshold. This prevents edge cases such as a score of 68 with `excessive_typos` and `garbage_characters` being silently accepted.
5. After the re-OCR pass, the fresh OCR result is compared **head-to-head** against the original embedded text using an AI side-by-side review. The higher-quality text is passed to downstream processing (metadata extraction, AI analysis). This ensures re-OCR never degrades quality.
6. All quality decisions (score, source, AI feedback, comparison outcome) are recorded in the processing log for review.
> **Tip**: Set `ENABLE_TEXT_QUALITY_CHECK=false` to disable the check entirely and always use embedded text as-is. This is useful when the AI provider is unavailable or when processing speed is more important than text accuracy. > **Tip**: Set `ENABLE_TEXT_QUALITY_CHECK=false` to disable the check entirely and always use embedded text as-is. This is useful when the AI provider is unavailable or when processing speed is more important than text accuracy.
> **Tuning the threshold**: The default of `TEXT_QUALITY_THRESHOLD=85` is intentionally strict. Lower it (e.g., `70`) for environments with consistently good existing OCR. Raise it (up to `100`) for maximum quality enforcement.
#### Searchable PDF Text Layer #### Searchable PDF Text Layer
Not all OCR providers embed a searchable text layer in the output PDF. The table below summarises each provider's behaviour and how DocuElevate handles it: Not all OCR providers embed a searchable text layer in the output PDF. The table below summarises each provider's behaviour and how DocuElevate handles it:
+338 -24
View File
@@ -4,6 +4,8 @@ Comprehensive tests for app/utils/text_quality.py.
Covers: Covers:
- detect_pdf_text_source: digital, OCR, and unknown PDF metadata - detect_pdf_text_source: digital, OCR, and unknown PDF metadata
- check_text_quality: digital bypass, good text, poor text, empty text - check_text_quality: digital bypass, good text, poor text, empty text
- Strict threshold (85) and significant-issues override logic
- compare_text_quality: head-to-head comparison between original and OCR text
- AI failure and JSON-parse error handling - AI failure and JSON-parse error handling
- Integration with process_document: quality check disabled, good quality, - Integration with process_document: quality check disabled, good quality,
poor quality (triggers re-OCR), digital source bypass poor quality (triggers re-OCR), digital source bypass
@@ -14,9 +16,11 @@ from unittest.mock import MagicMock, patch
import pytest import pytest
from app.utils.text_quality import ( from app.utils.text_quality import (
TextComparisonResult,
TextQualityResult, TextQualityResult,
TextSource, TextSource,
check_text_quality, check_text_quality,
compare_text_quality,
detect_pdf_text_source, detect_pdf_text_source,
) )
@@ -257,6 +261,18 @@ class TestCheckTextQualityAI:
provider.chat_completion.return_value = response provider.chat_completion.return_value = response
return provider return provider
def _mock_settings(self, mock_settings, threshold: int = 85):
"""Configure mock settings with sensible defaults for quality check tests."""
mock_settings.ai_model = "gpt-4o-mini"
mock_settings.openai_model = "gpt-4o-mini"
mock_settings.text_quality_threshold = threshold
mock_settings.text_quality_significant_issues = [
"excessive_typos",
"garbage_characters",
"incoherent_text",
"fragmented_sentences",
]
def test_good_text_passes(self): def test_good_text_passes(self):
"""A high-quality AI response marks text as good.""" """A high-quality AI response marks text as good."""
ai_resp = _make_ai_response(90, True, "Well-structured invoice text.", []) ai_resp = _make_ai_response(90, True, "Well-structured invoice text.", [])
@@ -266,8 +282,7 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = "gpt-4o-mini" self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS) result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is True assert result.is_good_quality is True
@@ -285,8 +300,7 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = "gpt-4o-mini" self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
result = check_text_quality(POOR_OCR_TEXT, TextSource.OCR_PREVIOUS) result = check_text_quality(POOR_OCR_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is False assert result.is_good_quality is False
@@ -303,8 +317,7 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = "gpt-4o-mini" self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
result = check_text_quality(GARBAGE_TEXT, TextSource.UNKNOWN) result = check_text_quality(GARBAGE_TEXT, TextSource.UNKNOWN)
assert result.is_good_quality is False assert result.is_good_quality is False
@@ -319,34 +332,125 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = None self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
result = check_text_quality(FRAGMENTED_TEXT, TextSource.OCR_PREVIOUS) result = check_text_quality(FRAGMENTED_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is False assert result.is_good_quality is False
def test_borderline_score_uses_is_good_quality_field(self): def test_score_below_threshold_rejected_even_if_ai_says_good(self):
"""The is_good_quality field from the AI takes precedence over the threshold.""" """Score below threshold forces rejection even when AI returns is_good_quality=True."""
# Score 64 but AI explicitly says True # This is the key issue from the bug report: score=68 with is_good_quality=True
ai_resp = _make_ai_response(64, True, "Mostly readable despite minor issues.", []) # should NOT be accepted.
ai_resp = _make_ai_response(68, True, "Text is largely legible with some OCR-induced typos.", [])
provider = self._mock_provider(ai_resp) provider = self._mock_provider(ai_resp)
with ( with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = None self._mock_settings(mock_settings, threshold=85)
mock_settings.openai_model = "gpt-4o-mini" result = check_text_quality(POOR_OCR_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is False
assert result.quality_score == 68
def test_significant_issues_force_rejection_above_threshold(self):
"""Significant issues force rejection even when score is above threshold."""
# Score is 88 (above default threshold of 85), but has excessive_typos and garbage_characters.
ai_resp = _make_ai_response(
88,
True,
"Mostly readable text with some OCR artefacts.",
["excessive_typos", "garbage_characters"],
)
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
self._mock_settings(mock_settings, threshold=85)
result = check_text_quality(POOR_OCR_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is False
assert result.quality_score == 88
def test_significant_issues_with_incoherent_text(self):
"""incoherent_text in issues forces rejection even above threshold."""
ai_resp = _make_ai_response(
90,
True,
"Text has coherent paragraphs but some incoherent passages.",
["incoherent_text"],
)
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
self._mock_settings(mock_settings, threshold=85)
result = check_text_quality(POOR_OCR_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is False
def test_non_significant_issues_do_not_block_good_score(self):
"""Issues not in significant_issues list do not override a good score."""
# 'minor_formatting' is not in the significant issues list.
ai_resp = _make_ai_response(
90,
True,
"Well-structured text with minor formatting issues.",
["minor_formatting"],
)
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
self._mock_settings(mock_settings, threshold=85)
result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS) result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is True assert result.is_good_quality is True
assert result.quality_score == 64 assert result.quality_score == 90
def test_borderline_score_below_threshold_is_rejected(self):
"""A score just below the threshold is rejected regardless of AI verdict."""
ai_resp = _make_ai_response(84, True, "Mostly readable despite minor issues.", [])
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
self._mock_settings(mock_settings, threshold=85)
result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS)
# Score 84 < threshold 85 → rejected
assert result.is_good_quality is False
assert result.quality_score == 84
def test_borderline_score_at_threshold_is_accepted(self):
"""A score exactly at the threshold is accepted when no significant issues."""
ai_resp = _make_ai_response(85, True, "Meets the quality threshold.", [])
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
self._mock_settings(mock_settings, threshold=85)
result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is True
assert result.quality_score == 85
def test_markdown_fences_stripped_before_parse(self): def test_markdown_fences_stripped_before_parse(self):
"""The parser handles AI responses wrapped in markdown code fences.""" """The parser handles AI responses wrapped in markdown code fences."""
import json as _json import json as _json
inner = _json.dumps({"quality_score": 80, "is_good_quality": True, "feedback": "Fine.", "issues": []}) inner = _json.dumps({"quality_score": 90, "is_good_quality": True, "feedback": "Fine.", "issues": []})
fenced = f"```json\n{inner}\n```" fenced = f"```json\n{inner}\n```"
provider = self._mock_provider(fenced) provider = self._mock_provider(fenced)
@@ -354,12 +458,11 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = None self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
result = check_text_quality(GOOD_TEXT, TextSource.UNKNOWN) result = check_text_quality(GOOD_TEXT, TextSource.UNKNOWN)
assert result.is_good_quality is True assert result.is_good_quality is True
assert result.quality_score == 80 assert result.quality_score == 90
def test_raw_ai_response_stored_in_result(self): def test_raw_ai_response_stored_in_result(self):
"""The raw AI response is preserved in TextQualityResult.ai_response_raw.""" """The raw AI response is preserved in TextQualityResult.ai_response_raw."""
@@ -370,8 +473,7 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = None self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS) result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS)
assert result.ai_response_raw == ai_resp assert result.ai_response_raw == ai_resp
@@ -381,7 +483,7 @@ class TestCheckTextQualityAI:
from app.utils.text_quality import _TEXT_SAMPLE_MAX_CHARS from app.utils.text_quality import _TEXT_SAMPLE_MAX_CHARS
long_text = "a" * (_TEXT_SAMPLE_MAX_CHARS + 5000) long_text = "a" * (_TEXT_SAMPLE_MAX_CHARS + 5000)
ai_resp = _make_ai_response(85, True, "Fine.", []) ai_resp = _make_ai_response(90, True, "Fine.", [])
provider = self._mock_provider(ai_resp) provider = self._mock_provider(ai_resp)
captured_prompts: list[str] = [] captured_prompts: list[str] = []
@@ -395,8 +497,7 @@ class TestCheckTextQualityAI:
patch("app.utils.text_quality.get_ai_provider", return_value=provider), patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings, patch("app.utils.text_quality.settings") as mock_settings,
): ):
mock_settings.ai_model = None self._mock_settings(mock_settings)
mock_settings.openai_model = "gpt-4o-mini"
check_text_quality(long_text, TextSource.UNKNOWN) check_text_quality(long_text, TextSource.UNKNOWN)
assert len(captured_prompts) == 1 assert len(captured_prompts) == 1
@@ -424,6 +525,13 @@ class TestCheckTextQualityErrorHandling:
): ):
mock_settings.ai_model = None mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini" mock_settings.openai_model = "gpt-4o-mini"
mock_settings.text_quality_threshold = 85
mock_settings.text_quality_significant_issues = [
"excessive_typos",
"garbage_characters",
"incoherent_text",
"fragmented_sentences",
]
result = check_text_quality(GOOD_TEXT, TextSource.UNKNOWN) result = check_text_quality(GOOD_TEXT, TextSource.UNKNOWN)
assert result.is_good_quality is True assert result.is_good_quality is True
@@ -440,6 +548,13 @@ class TestCheckTextQualityErrorHandling:
): ):
mock_settings.ai_model = None mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini" mock_settings.openai_model = "gpt-4o-mini"
mock_settings.text_quality_threshold = 85
mock_settings.text_quality_significant_issues = [
"excessive_typos",
"garbage_characters",
"incoherent_text",
"fragmented_sentences",
]
result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS) result = check_text_quality(GOOD_TEXT, TextSource.OCR_PREVIOUS)
assert result.is_good_quality is True assert result.is_good_quality is True
@@ -456,6 +571,13 @@ class TestCheckTextQualityErrorHandling:
): ):
mock_settings.ai_model = None mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini" mock_settings.openai_model = "gpt-4o-mini"
mock_settings.text_quality_threshold = 85
mock_settings.text_quality_significant_issues = [
"excessive_typos",
"garbage_characters",
"incoherent_text",
"fragmented_sentences",
]
result = check_text_quality(GOOD_TEXT, TextSource.UNKNOWN) result = check_text_quality(GOOD_TEXT, TextSource.UNKNOWN)
assert result.ai_response_raw is None assert result.ai_response_raw is None
@@ -665,6 +787,13 @@ class TestProcessDocumentTextQuality:
mock_ocr.delay.assert_called_once() mock_ocr.delay.assert_called_once()
mock_gpt.delay.assert_not_called() mock_gpt.delay.assert_not_called()
assert "OCR" in result["status"] assert "OCR" in result["status"]
# Verify original text is passed to the OCR task for comparison
call_args = mock_ocr.delay.call_args
assert call_args is not None
# Third argument (original_text) should be a non-empty string
assert len(call_args.args) >= 3
assert isinstance(call_args.args[2], str)
assert len(call_args.args[2]) > 0
def test_quality_check_digital_source_skips_ai_call(self, db_session, tmp_path): def test_quality_check_digital_source_skips_ai_call(self, db_session, tmp_path):
"""Digital-origin PDFs bypass the AI and proceed directly to GPT.""" """Digital-origin PDFs bypass the AI and proceed directly to GPT."""
@@ -700,3 +829,188 @@ class TestProcessDocumentTextQuality:
mock_gpt.delay.assert_called_once() mock_gpt.delay.assert_called_once()
mock_ocr.delay.assert_not_called() mock_ocr.delay.assert_not_called()
assert result["status"] == "Text extracted locally" assert result["status"] == "Text extracted locally"
# ---------------------------------------------------------------------------
# TextComparisonResult dataclass
# ---------------------------------------------------------------------------
@pytest.mark.unit
class TestTextComparisonResult:
"""Tests for the TextComparisonResult dataclass."""
def test_default_ai_response_raw_is_none(self):
result = TextComparisonResult(
preferred="ocr",
original_score=60,
ocr_score=85,
explanation="OCR is cleaner.",
)
assert result.preferred == "ocr"
assert result.original_score == 60
assert result.ocr_score == 85
assert result.ai_response_raw is None
def test_preferred_original(self):
result = TextComparisonResult(
preferred="original",
original_score=90,
ocr_score=70,
explanation="Original is higher quality.",
)
assert result.preferred == "original"
# ---------------------------------------------------------------------------
# compare_text_quality
# ---------------------------------------------------------------------------
def _make_comparison_response(original_score: int, ocr_score: int, preferred: str, explanation: str) -> str:
import json as _json
return _json.dumps(
{
"original_score": original_score,
"ocr_score": ocr_score,
"preferred": preferred,
"explanation": explanation,
}
)
@pytest.mark.unit
class TestCompareTextQuality:
"""Tests for compare_text_quality() head-to-head comparison."""
def _mock_provider(self, response: str) -> MagicMock:
provider = MagicMock()
provider.chat_completion.return_value = response
return provider
def test_ocr_preferred(self):
"""When AI prefers OCR text, preferred='ocr'."""
ai_resp = _make_comparison_response(60, 88, "ocr", "OCR output is much cleaner.")
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = "gpt-4o-mini"
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(POOR_OCR_TEXT, GOOD_TEXT)
assert result.preferred == "ocr"
assert result.original_score == 60
assert result.ocr_score == 88
assert "OCR" in result.explanation
def test_original_preferred(self):
"""When AI prefers original text, preferred='original'."""
ai_resp = _make_comparison_response(92, 70, "original", "Original is higher quality.")
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = "gpt-4o-mini"
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(GOOD_TEXT, POOR_OCR_TEXT)
assert result.preferred == "original"
assert result.original_score == 92
assert result.ocr_score == 70
def test_equal_preferred(self):
"""When AI finds both equal, preferred='equal'."""
ai_resp = _make_comparison_response(85, 85, "equal", "Both texts are equivalent quality.")
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(GOOD_TEXT, GOOD_TEXT)
assert result.preferred == "equal"
def test_invalid_preferred_value_defaults_to_ocr(self):
"""An unexpected preferred value in the AI response defaults to 'ocr'."""
import json as _json
ai_resp = _json.dumps(
{
"original_score": 80,
"ocr_score": 75,
"preferred": "neither", # invalid
"explanation": "Both are bad.",
}
)
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(GOOD_TEXT, GOOD_TEXT)
assert result.preferred == "ocr"
def test_markdown_fences_stripped(self):
"""Markdown code fences in comparison response are stripped before parsing."""
import json as _json
inner = _json.dumps(
{"original_score": 70, "ocr_score": 90, "preferred": "ocr", "explanation": "OCR is better."}
)
fenced = f"```json\n{inner}\n```"
provider = self._mock_provider(fenced)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(POOR_OCR_TEXT, GOOD_TEXT)
assert result.preferred == "ocr"
assert result.ocr_score == 90
def test_ai_error_defaults_to_ocr(self):
"""AI error during comparison safely defaults to OCR text preferred."""
provider = MagicMock()
provider.chat_completion.side_effect = RuntimeError("Timeout")
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(POOR_OCR_TEXT, GOOD_TEXT)
assert result.preferred == "ocr"
assert "Timeout" in result.explanation
def test_raw_response_stored(self):
"""The raw AI response is stored in ai_response_raw."""
ai_resp = _make_comparison_response(75, 88, "ocr", "OCR is better.")
provider = self._mock_provider(ai_resp)
with (
patch("app.utils.text_quality.get_ai_provider", return_value=provider),
patch("app.utils.text_quality.settings") as mock_settings,
):
mock_settings.ai_model = None
mock_settings.openai_model = "gpt-4o-mini"
result = compare_text_quality(POOR_OCR_TEXT, GOOD_TEXT)
assert result.ai_response_raw == ai_resp