fix(tasks): skip duplicate check when reprocessing and enable retry from failed pipeline step

- Add file_id parameter to process_document to skip duplicate hash check on reprocess
- Pass file_id from reprocess_single_file and bulk_reprocess_files endpoints
- Extend retry-subtask endpoint to support pipeline steps (process_document,
  process_with_azure_document_intelligence, extract_metadata_with_gpt,
  embed_metadata_into_pdf) in addition to upload tasks
- Add retry button for failed main pipeline steps in file detail UI
- Add comprehensive tests for reprocessing and pipeline step retry

Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
copilot-swe-agent[bot]
2026-02-10 21:09:53 +00:00
parent 7c074c2755
commit 11c7d15a90
6 changed files with 484 additions and 58 deletions
+8 -2
View File
@@ -2,10 +2,12 @@
Tests for bulk file operations (delete and reprocess).
"""
from unittest.mock import MagicMock, patch
import pytest
from fastapi.testclient import TestClient
from app.models import FileRecord, ProcessingLog
from unittest.mock import patch, MagicMock
@pytest.mark.integration
@@ -94,7 +96,7 @@ class TestBulkOperations:
@patch("app.api.files.process_document")
def test_bulk_reprocess_success(self, mock_process_document, client: TestClient, db_session):
"""Test bulk reprocessing of files."""
"""Test bulk reprocessing of files passes file_id to skip duplicate check."""
# Setup mock
mock_task = MagicMock()
mock_task.id = "test-task-id"
@@ -125,6 +127,10 @@ class TestBulkOperations:
assert len(data["processed_files"]) == 2
assert len(data["task_ids"]) == 2
# Verify that file_id was passed to skip duplicate check
for call_args in mock_process_document.delay.call_args_list:
assert "file_id" in call_args.kwargs or len(call_args.args) > 1
@patch("app.api.files.process_document")
def test_bulk_reprocess_missing_files(self, mock_process_document, client: TestClient, db_session):
"""Test bulk reprocessing when some local files are missing."""
+124 -1
View File
@@ -3,9 +3,11 @@ Tests for file detail view improvements including reprocessing and preview endpo
"""
import os
from unittest.mock import MagicMock, patch
import pytest
from unittest.mock import patch, MagicMock
from fastapi.testclient import TestClient
from app.models import FileRecord, ProcessingLog
@@ -127,6 +129,127 @@ class TestSubtaskRetry:
assert response.status_code == 400
assert "processed file not found" in response.json()["detail"].lower()
@patch("app.api.files.process_document")
def test_retry_pipeline_step_process_document(
self, mock_process_document, client: TestClient, db_session, sample_pdf_path
):
"""Test retrying the process_document pipeline step."""
mock_task = MagicMock()
mock_task.id = "retry-task-123"
mock_process_document.delay.return_value = mock_task
file_record = FileRecord(
filehash="pipeline_retry1",
original_filename="pipeline.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf",
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
response = client.post(f"/api/files/{file_record.id}/retry-subtask?subtask_name=process_document")
assert response.status_code == 200
data = response.json()
assert data["status"] == "success"
assert data["subtask_name"] == "process_document"
assert "task_id" in data
# Verify process_document.delay was called with file_id
mock_process_document.delay.assert_called_once()
call_kwargs = mock_process_document.delay.call_args
assert call_kwargs[1].get("file_id") == file_record.id or call_kwargs[0][-1] == file_record.id
def test_retry_pipeline_step_ocr(self, client: TestClient, db_session, sample_pdf_path):
"""Test retrying the OCR pipeline step."""
mock_task = MagicMock()
mock_task.id = "ocr-retry-task"
file_record = FileRecord(
filehash="pipeline_retry2",
original_filename="ocr_retry.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf",
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
with patch(
"app.tasks.process_with_azure_document_intelligence.process_with_azure_document_intelligence"
) as mock_azure:
mock_azure.delay.return_value = mock_task
response = client.post(
f"/api/files/{file_record.id}/retry-subtask?subtask_name=process_with_azure_document_intelligence"
)
assert response.status_code == 200
data = response.json()
assert data["status"] == "success"
assert data["subtask_name"] == "process_with_azure_document_intelligence"
def test_retry_pipeline_step_metadata_extraction(self, client: TestClient, db_session, sample_pdf_path):
"""Test retrying the metadata extraction pipeline step."""
mock_task = MagicMock()
mock_task.id = "gpt-retry-task"
file_record = FileRecord(
filehash="pipeline_retry3",
original_filename="gpt_retry.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf",
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
with patch("app.tasks.extract_metadata_with_gpt.extract_metadata_with_gpt") as mock_gpt:
mock_gpt.delay.return_value = mock_task
response = client.post(f"/api/files/{file_record.id}/retry-subtask?subtask_name=extract_metadata_with_gpt")
assert response.status_code == 200
data = response.json()
assert data["status"] == "success"
assert data["subtask_name"] == "extract_metadata_with_gpt"
def test_retry_pipeline_step_embed_no_metadata(self, client: TestClient, db_session, sample_pdf_path):
"""Test retrying embed_metadata_into_pdf without prior metadata extraction."""
file_record = FileRecord(
filehash="pipeline_retry4",
original_filename="embed_retry.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf",
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Try to retry embed without a successful metadata extraction log
response = client.post(f"/api/files/{file_record.id}/retry-subtask?subtask_name=embed_metadata_into_pdf")
assert response.status_code == 400
assert "retry extract_metadata_with_gpt first" in response.json()["detail"].lower()
def test_retry_pipeline_step_missing_local_file(self, client: TestClient, db_session):
"""Test retrying a pipeline step when local file is missing."""
file_record = FileRecord(
filehash="pipeline_retry5",
original_filename="missing.pdf",
local_filename="/nonexistent/path/missing.pdf",
file_size=1024,
mime_type="application/pdf",
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
response = client.post(
f"/api/files/{file_record.id}/retry-subtask?subtask_name=process_with_azure_document_intelligence"
)
assert response.status_code == 400
assert "not found on disk" in response.json()["detail"].lower()
@pytest.mark.integration
class TestFilePreview:
+157 -14
View File
@@ -6,12 +6,13 @@ and doesn't cause DetachedInstanceError when accessing database objects.
"""
import os
from unittest.mock import MagicMock, patch
import pytest
from unittest.mock import patch, MagicMock
from sqlalchemy.orm import Session
from app.tasks.process_document import process_document
from app.models import FileRecord
from app.tasks.process_document import process_document
@pytest.mark.unit
@@ -85,11 +86,12 @@ startxref
test_pdf.write_bytes(pdf_content)
# Mock environment and dependencies
with patch("app.tasks.process_document.SessionLocal") as mock_session_local, patch(
"app.tasks.process_document.settings"
) as mock_settings, patch("app.tasks.process_document.log_task_progress"), patch(
"app.tasks.process_document.extract_metadata_with_gpt"
) as mock_extract:
with (
patch("app.tasks.process_document.SessionLocal") as mock_session_local,
patch("app.tasks.process_document.settings") as mock_settings,
patch("app.tasks.process_document.log_task_progress"),
patch("app.tasks.process_document.extract_metadata_with_gpt") as mock_extract,
):
# Setup mocks
mock_settings.workdir = str(tmp_path)
@@ -147,8 +149,9 @@ def test_process_document_duplicate_file(db_session, tmp_path):
existing_id = existing_record.id
# Mock environment and dependencies
with patch("app.tasks.process_document.SessionLocal") as mock_session_local, patch(
"app.tasks.process_document.log_task_progress"
with (
patch("app.tasks.process_document.SessionLocal") as mock_session_local,
patch("app.tasks.process_document.log_task_progress"),
):
# Setup mocks
@@ -213,11 +216,12 @@ startxref
test_pdf.write_bytes(pdf_content)
# Mock environment and dependencies
with patch("app.tasks.process_document.SessionLocal") as mock_session_local, patch(
"app.tasks.process_document.settings"
) as mock_settings, patch("app.tasks.process_document.log_task_progress"), patch(
"app.tasks.process_document.process_with_azure_document_intelligence"
) as mock_azure:
with (
patch("app.tasks.process_document.SessionLocal") as mock_session_local,
patch("app.tasks.process_document.settings") as mock_settings,
patch("app.tasks.process_document.log_task_progress"),
patch("app.tasks.process_document.process_with_azure_document_intelligence") as mock_azure,
):
# Setup mocks
mock_settings.workdir = str(tmp_path)
@@ -242,3 +246,142 @@ startxref
# The second argument should be the file_id
assert call_args[0][1] == file_record.id
@pytest.mark.unit
@pytest.mark.requires_db
def test_process_document_reprocess_skips_duplicate_check(db_session, tmp_path):
"""
Test that reprocessing an existing file (with file_id) skips the duplicate check
and continues processing normally.
"""
# Create a test PDF file with embedded text
test_pdf = tmp_path / "test.pdf"
pdf_content = b"""%PDF-1.4
1 0 obj
<<
/Type /Catalog
/Pages 2 0 R
>>
endobj
2 0 obj
<<
/Type /Pages
/Kids [3 0 R]
/Count 1
>>
endobj
3 0 obj
<<
/Type /Page
/Parent 2 0 R
/MediaBox [0 0 612 792]
/Resources <<
/Font <<
/F1 <<
/Type /Font
/Subtype /Type1
/BaseFont /Helvetica
>>
>>
>>
/Contents 4 0 R
>>
endobj
4 0 obj
<<
/Length 44
>>
stream
BT
/F1 12 Tf
100 700 Td
(Test content) Tj
ET
endstream
endobj
xref
0 5
0000000000 65535 f
0000000009 00000 n
0000000058 00000 n
0000000115 00000 n
0000000306 00000 n
trailer
<<
/Size 5
/Root 1 0 R
>>
startxref
399
%%EOF
"""
test_pdf.write_bytes(pdf_content)
# Pre-create a FileRecord with the same hash (simulating an existing record)
from app.utils import hash_file
filehash = hash_file(str(test_pdf))
existing_record = FileRecord(
filehash=filehash,
original_filename="test.pdf",
local_filename=str(test_pdf),
file_size=len(pdf_content),
mime_type="application/pdf",
)
db_session.add(existing_record)
db_session.commit()
existing_id = existing_record.id
# Mock environment and dependencies
with (
patch("app.tasks.process_document.SessionLocal") as mock_session_local,
patch("app.tasks.process_document.settings") as mock_settings,
patch("app.tasks.process_document.log_task_progress"),
patch("app.tasks.process_document.extract_metadata_with_gpt") as mock_extract,
):
# Setup mocks
mock_settings.workdir = str(tmp_path)
mock_session_local.return_value.__enter__.return_value = db_session
mock_session_local.return_value.__exit__.return_value = None
mock_extract.delay = MagicMock()
# Call with file_id to trigger reprocessing (should skip duplicate check)
result = process_document.run(str(test_pdf), file_id=existing_id)
# Verify that processing continued (not blocked by duplicate check)
assert result["status"] == "Text extracted locally"
assert result["file_id"] == existing_id
# Verify that extract_metadata_with_gpt was called
mock_extract.delay.assert_called_once()
# Verify that only one FileRecord still exists (no new record created)
assert db_session.query(FileRecord).count() == 1
@pytest.mark.unit
@pytest.mark.requires_db
def test_process_document_reprocess_nonexistent_file_id(db_session, tmp_path):
"""
Test that reprocessing with a non-existent file_id returns an error.
"""
# Create a test PDF file
test_pdf = tmp_path / "test.pdf"
test_pdf.write_bytes(b"test content")
with (
patch("app.tasks.process_document.SessionLocal") as mock_session_local,
patch("app.tasks.process_document.log_task_progress"),
):
mock_session_local.return_value.__enter__.return_value = db_session
mock_session_local.return_value.__exit__.return_value = None
# Call with a file_id that doesn't exist
result = process_document.run(str(test_pdf), file_id=99999)
# Verify error is returned
assert "error" in result
assert result["file_id"] == 99999