fix(status): prevent false Completed status when mandatory pipeline steps have not run

Add a terminal-step guard (send_to_all_destinations) to all status
calculation paths so that files are only marked Completed once the
entire processing pipeline has been recorded.

- get_file_overall_status: require TERMINAL_STEP to be present
- get_files_processing_status: same guard for bulk status
- get_step_summary: count missing terminal step as queued so
  total_main_steps > main_completed when pipeline is incomplete
- apply_status_filter: SQL sub-query requires terminal step for
  completed filter
- process_document: call initialize_file_steps after creating a new
  file record so all mandatory steps are pre-created as pending

Define TERMINAL_STEP constant in step_manager.py and reference it in
file_status.py and file_queries.py to avoid magic strings.

Tests updated: add send_to_all_destinations to completed-file
fixtures; add test verifying initialize_file_steps is called for
new files.

Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
copilot-swe-agent[bot]
2026-02-27 00:51:14 +00:00
parent 515f5b97e9
commit e6dd39c27d
7 changed files with 165 additions and 12 deletions
+99
View File
@@ -743,3 +743,102 @@ startxref
# Verify that retry was triggered
# The retry method raises a special exception
assert exc_info.value is not None
@pytest.mark.unit
@pytest.mark.requires_db
def test_process_document_initializes_file_steps_for_new_file(db_session, tmp_path):
"""
Test that process_document calls initialize_file_steps for new file records
so that all mandatory pipeline steps are pre-created as "pending".
This ensures status tracking reflects the complete expected pipeline from
the start and prevents incomplete files from being falsely marked as
"completed" just because the steps that *did* run all succeeded.
"""
# Create a test PDF file with embedded text
test_pdf = tmp_path / "test.pdf"
pdf_content = b"""%PDF-1.4
1 0 obj
<<
/Type /Catalog
/Pages 2 0 R
>>
endobj
2 0 obj
<<
/Type /Pages
/Kids [3 0 R]
/Count 1
>>
endobj
3 0 obj
<<
/Type /Page
/Parent 2 0 R
/MediaBox [0 0 612 792]
/Resources <<
/Font <<
/F1 <<
/Type /Font
/Subtype /Type1
/BaseFont /Helvetica
>>
>>
>>
/Contents 4 0 R
>>
endobj
4 0 obj
<<
/Length 44
>>
stream
BT
/F1 12 Tf
100 700 Td
(Test content) Tj
ET
endstream
endobj
xref
0 5
0000000000 65535 f
0000000009 00000 n
0000000058 00000 n
0000000115 00000 n
0000000306 00000 n
trailer
<<
/Size 5
/Root 1 0 R
>>
startxref
399
%%EOF
"""
test_pdf.write_bytes(pdf_content)
with (
patch("app.tasks.process_document.SessionLocal") as mock_session_local,
patch("app.tasks.process_document.settings") as mock_settings,
patch("app.tasks.process_document.log_task_progress"),
patch("app.tasks.process_document.extract_metadata_with_gpt") as mock_extract,
patch("app.tasks.process_document.initialize_file_steps") as mock_init_steps,
):
mock_settings.workdir = str(tmp_path)
mock_settings.enable_deduplication = False
mock_settings.enable_text_quality_check = False
mock_session_local.return_value.__enter__.return_value = db_session
mock_session_local.return_value.__exit__.return_value = None
mock_extract.delay = MagicMock()
result = process_document.run(str(test_pdf))
assert result["status"] == "Text extracted locally"
assert "file_id" in result
# initialize_file_steps must have been called exactly once with the new file's ID
mock_init_steps.assert_called_once()
called_file_id = mock_init_steps.call_args[0][1]
assert called_file_id == result["file_id"]