Merge pull request #429 from christianlouis/copilot/fix-complete-status-error

fix(tests): align "completed" test fixtures with terminal-step guard semantics
This commit is contained in:
Christian Krakau-Louis
2026-02-27 10:22:39 +01:00
committed by GitHub
10 changed files with 195 additions and 38 deletions
+4
View File
@@ -17,6 +17,7 @@ from app.tasks.extract_metadata_with_gpt import extract_metadata_with_gpt
from app.tasks.process_with_ocr import process_with_ocr
from app.tasks.retry_config import BaseTaskWithRetry
from app.utils import get_unique_filepath_with_counter, hash_file, log_task_progress
from app.utils.step_manager import initialize_file_steps
from app.utils.text_quality import check_text_quality, detect_pdf_text_source
logger = logging.getLogger(__name__)
@@ -202,6 +203,9 @@ def process_document(
db.commit()
db.refresh(new_record)
logger.info(f"[{task_id}] File record created with ID: {new_record.id}")
# Pre-initialize all expected processing steps as "pending" so that
# status tracking reflects the complete pipeline from the start.
initialize_file_steps(db, new_record.id)
log_task_progress(
task_id,
"create_file_record",
+17 -3
View File
@@ -11,6 +11,7 @@ from sqlalchemy import or_
from sqlalchemy.orm import Query, Session
from app.models import FileProcessingStep, FileRecord
from app.utils.step_manager import TERMINAL_STEP
def apply_status_filter(query: Query, db: Session, status: Optional[str]) -> Query:
@@ -98,6 +99,9 @@ def apply_status_filter(query: Query, db: Session, status: Optional[str]) -> Que
query = query.filter(FileRecord.id.in_(db.query(subq.c.file_id)))
elif status == "completed":
# Files where all real steps are either success or skipped (no failures or in_progress)
# and the terminal send_to_all_destinations step has been recorded.
# Excluding the terminal-step requirement allows files that only completed the
# first few pipeline stages to be falsely labelled as "completed".
# Exclude duplicates from completed
query = query.filter(FileRecord.is_duplicate.is_(False))
@@ -113,9 +117,19 @@ def apply_status_filter(query: Query, db: Session, status: Optional[str]) -> Que
.subquery()
)
# Select files with real steps that don't have issues
query = query.filter(FileRecord.id.in_(db.query(files_with_real_steps.c.file_id))).filter(
~FileRecord.id.in_(db.query(files_with_issues.c.file_id))
# Get files that have the terminal processing step recorded
files_with_terminal_step = (
db.query(FileProcessingStep.file_id)
.filter(FileProcessingStep.step_name == TERMINAL_STEP)
.distinct()
.subquery()
)
# Select files with real steps, no issues, and terminal step present
query = (
query.filter(FileRecord.id.in_(db.query(files_with_real_steps.c.file_id)))
.filter(~FileRecord.id.in_(db.query(files_with_issues.c.file_id)))
.filter(FileRecord.id.in_(db.query(files_with_terminal_step.c.file_id)))
)
elif status == "duplicate":
# Files marked as duplicates
+9 -2
View File
@@ -7,7 +7,7 @@ from typing import Dict, List
from sqlalchemy.orm import Session
from app.models import FileProcessingStep, FileRecord, ProcessingLog
from app.utils.step_manager import get_file_overall_status
from app.utils.step_manager import TERMINAL_STEP, get_file_overall_status
def get_file_processing_status(db: Session, file_id: int) -> Dict:
@@ -150,7 +150,14 @@ def get_files_processing_status(db: Session, file_ids: List[int]) -> Dict[int, D
elif in_progress_steps > 0:
status = "processing"
elif completed_steps + skipped_steps == total_steps:
status = "completed"
# Only mark as completed if the terminal processing step has been
# recorded. Without this guard, files where later pipeline steps
# have not yet started would be falsely marked as "completed".
existing_step_names = {s.step_name for s in file_steps}
if TERMINAL_STEP in existing_step_names:
status = "completed"
else:
status = "pending"
else:
status = "pending"
+23 -1
View File
@@ -29,6 +29,10 @@ OPTIONAL_PROCESSING_STEPS = {
"check_for_duplicates": settings.enable_deduplication, # Only if deduplication is enabled
}
# The terminal step is the last mandatory step in the processing pipeline.
# A file is only considered "completed" once this step has been recorded.
TERMINAL_STEP = "send_to_all_destinations"
# Combine steps based on configuration
MAIN_PROCESSING_STEPS = []
if settings.enable_deduplication:
@@ -252,7 +256,14 @@ def get_file_overall_status(db: Session, file_id: int) -> Dict:
elif in_progress_steps > 0:
status = "processing"
elif completed_steps + skipped_steps == total_steps:
status = "completed"
# Only mark as completed if the terminal processing step has been recorded.
# Without this guard, files where later pipeline steps have not yet started
# would be falsely marked as "completed" (e.g. only the first 3 steps ran).
existing_step_names = {s.step_name for s in steps}
if TERMINAL_STEP in existing_step_names:
status = "completed"
else:
status = "pending"
else:
status = "pending"
@@ -340,6 +351,17 @@ def get_step_summary(db: Session, file_id: int) -> Dict:
main_counts[status] += 1
main_steps_count += 1
# Ensure the terminal step is always counted in total_main_steps.
# If the terminal step has not been recorded yet, the pipeline is not
# complete; counting it as "queued" prevents the status banner from
# showing "Completed" before the full pipeline has run.
# Note: if TERMINAL_STEP already appears in `steps`, the loop above has
# already incremented `main_steps_count` for it, so we only add here when
# the step is absent from the DB entirely.
if not any(step.step_name == TERMINAL_STEP for step in steps):
main_steps_count += 1
main_counts["queued"] += 1
return {
"main": main_counts,
"uploads": upload_counts,