81a04794f9
Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
821 lines
25 KiB
Python
821 lines
25 KiB
Python
"""
|
|
Unit tests for OCR document processing tasks.
|
|
|
|
These tests verify OCR processing logic with mocked external AI/ML services
|
|
(OpenAI, Azure Document Intelligence). Tests cover typical and edge cases.
|
|
"""
|
|
|
|
import os
|
|
import pytest
|
|
from unittest.mock import patch, MagicMock, Mock
|
|
from azure.ai.documentintelligence.models import AnalyzeResult
|
|
|
|
from app.tasks.process_with_azure_document_intelligence import (
|
|
process_with_azure_document_intelligence,
|
|
get_pdf_page_count,
|
|
check_page_rotation,
|
|
AZURE_DOC_INTELLIGENCE_LIMITS,
|
|
)
|
|
from app.tasks.refine_text_with_gpt import refine_text_with_gpt
|
|
from app.tasks.rotate_pdf_pages import rotate_pdf_pages, determine_rotation_angle
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestProcessWithAzureDocumentIntelligence:
|
|
"""Tests for Azure Document Intelligence OCR processing."""
|
|
|
|
def test_successful_ocr_processing(self, tmp_path):
|
|
"""Test successful OCR processing with Azure Document Intelligence."""
|
|
# Create tmp directory and test PDF file
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
pdf_content = b"""%PDF-1.4
|
|
1 0 obj
|
|
<<
|
|
/Type /Catalog
|
|
/Pages 2 0 R
|
|
>>
|
|
endobj
|
|
2 0 obj
|
|
<<
|
|
/Type /Pages
|
|
/Kids [3 0 R]
|
|
/Count 1
|
|
>>
|
|
endobj
|
|
3 0 obj
|
|
<<
|
|
/Type /Page
|
|
/Parent 2 0 R
|
|
/MediaBox [0 0 612 792]
|
|
>>
|
|
endobj
|
|
xref
|
|
0 4
|
|
0000000000 65535 f
|
|
0000000009 00000 n
|
|
0000000058 00000 n
|
|
0000000115 00000 n
|
|
trailer
|
|
<<
|
|
/Size 4
|
|
/Root 1 0 R
|
|
>>
|
|
startxref
|
|
197
|
|
%%EOF
|
|
"""
|
|
test_pdf.write_bytes(pdf_content)
|
|
|
|
# Mock Azure Document Intelligence client and response
|
|
mock_page = Mock()
|
|
mock_page.angle = 0
|
|
|
|
mock_result = Mock(spec=AnalyzeResult)
|
|
mock_result.content = "This is extracted text from OCR"
|
|
mock_result.model_id = "prebuilt-read"
|
|
mock_result.pages = [mock_page]
|
|
|
|
mock_poller = Mock()
|
|
mock_poller.result.return_value = mock_result
|
|
mock_poller.details = {"operation_id": "test-operation-123"}
|
|
|
|
mock_client = Mock()
|
|
mock_client.begin_analyze_document.return_value = mock_poller
|
|
mock_client.get_analyze_result_pdf.return_value = iter([b"searchable pdf content"])
|
|
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.document_intelligence_client",
|
|
mock_client,
|
|
), patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.process_with_azure_document_intelligence.rotate_pdf_pages"
|
|
) as mock_rotate:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_rotate.delay = MagicMock()
|
|
|
|
# Run the task
|
|
result = process_with_azure_document_intelligence.run(
|
|
filename=test_pdf.name, file_id=1
|
|
)
|
|
|
|
# Verify results
|
|
assert result["file"] == test_pdf.name
|
|
assert result["cleaned_text"] == "This is extracted text from OCR"
|
|
assert "searchable_pdf" in result
|
|
|
|
# Verify Azure client was called correctly
|
|
mock_client.begin_analyze_document.assert_called_once()
|
|
mock_client.get_analyze_result_pdf.assert_called_once_with(
|
|
model_id="prebuilt-read", result_id="test-operation-123"
|
|
)
|
|
|
|
# Verify rotate_pdf_pages was queued
|
|
mock_rotate.delay.assert_called_once()
|
|
call_args = mock_rotate.delay.call_args
|
|
assert call_args[0][0] == test_pdf.name
|
|
assert call_args[0][1] == "This is extracted text from OCR"
|
|
assert call_args[0][3] == 1 # file_id
|
|
|
|
def test_file_not_found_error(self, tmp_path):
|
|
"""Test that FileNotFoundError is raised when file doesn't exist."""
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings:
|
|
mock_settings.workdir = str(tmp_path)
|
|
|
|
with pytest.raises(FileNotFoundError) as exc_info:
|
|
process_with_azure_document_intelligence.run(
|
|
filename="nonexistent.pdf", file_id=1
|
|
)
|
|
|
|
assert "Local file not found" in str(exc_info.value)
|
|
|
|
def test_file_size_limit_exceeded(self, tmp_path):
|
|
"""Test file size limit validation (500 MB)."""
|
|
# Create tmp directory and test PDF file
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "large.pdf"
|
|
test_pdf.write_bytes(b"dummy content")
|
|
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.process_with_azure_document_intelligence.os.path.getsize"
|
|
) as mock_getsize:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
# Mock file size to be larger than 500 MB
|
|
mock_getsize.return_value = AZURE_DOC_INTELLIGENCE_LIMITS[
|
|
"max_file_size_bytes"
|
|
] + 1024
|
|
|
|
result = process_with_azure_document_intelligence.run(
|
|
filename=test_pdf.name, file_id=1
|
|
)
|
|
|
|
# Verify error response
|
|
assert "error" in result
|
|
assert "Size limit exceeded" in result["status"]
|
|
assert "500 MB" in result["error"]
|
|
|
|
def test_page_count_limit_exceeded(self, tmp_path):
|
|
"""Test page count limit validation (2000 pages)."""
|
|
# Create tmp directory and test PDF file
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "many_pages.pdf"
|
|
test_pdf.write_bytes(b"%PDF-1.4\n%%EOF") # Minimal valid PDF
|
|
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.process_with_azure_document_intelligence.get_pdf_page_count"
|
|
) as mock_page_count:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
# Mock page count to exceed limit
|
|
mock_page_count.return_value = AZURE_DOC_INTELLIGENCE_LIMITS["max_pages"] + 1
|
|
|
|
result = process_with_azure_document_intelligence.run(
|
|
filename=test_pdf.name, file_id=1
|
|
)
|
|
|
|
# Verify error response
|
|
assert "error" in result
|
|
assert "Page limit exceeded" in result["status"]
|
|
assert "2000 pages" in result["error"]
|
|
|
|
def test_page_count_unknown_proceeds_with_processing(self, tmp_path):
|
|
"""Test that processing continues when page count cannot be determined."""
|
|
# Create tmp directory and test PDF file
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
test_pdf.write_bytes(b"%PDF-1.4\n%%EOF")
|
|
|
|
mock_page = Mock()
|
|
mock_page.angle = 0
|
|
|
|
mock_result = Mock(spec=AnalyzeResult)
|
|
mock_result.content = "Test content"
|
|
mock_result.model_id = "prebuilt-read"
|
|
mock_result.pages = [mock_page]
|
|
|
|
mock_poller = Mock()
|
|
mock_poller.result.return_value = mock_result
|
|
mock_poller.details = {"operation_id": "test-123"}
|
|
|
|
mock_client = Mock()
|
|
mock_client.begin_analyze_document.return_value = mock_poller
|
|
mock_client.get_analyze_result_pdf.return_value = iter([b"content"])
|
|
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.document_intelligence_client",
|
|
mock_client,
|
|
), patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.process_with_azure_document_intelligence.get_pdf_page_count"
|
|
) as mock_page_count, patch(
|
|
"app.tasks.process_with_azure_document_intelligence.rotate_pdf_pages"
|
|
):
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
# Return None to simulate page count determination failure
|
|
mock_page_count.return_value = None
|
|
|
|
# Should not raise an error
|
|
result = process_with_azure_document_intelligence.run(
|
|
filename=test_pdf.name, file_id=1
|
|
)
|
|
|
|
# Verify processing continued
|
|
assert "error" not in result
|
|
assert result["file"] == test_pdf.name
|
|
|
|
def test_page_rotation_detection(self, tmp_path):
|
|
"""Test detection and logging of page rotation."""
|
|
# Create tmp directory and test PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "rotated.pdf"
|
|
test_pdf.write_bytes(b"%PDF-1.4\n%%EOF")
|
|
|
|
# Mock pages with different rotation angles
|
|
mock_page1 = Mock()
|
|
mock_page1.angle = 90
|
|
|
|
mock_page2 = Mock()
|
|
mock_page2.angle = 0
|
|
|
|
mock_page3 = Mock()
|
|
mock_page3.angle = 180
|
|
|
|
mock_result = Mock(spec=AnalyzeResult)
|
|
mock_result.content = "Rotated document text"
|
|
mock_result.model_id = "prebuilt-read"
|
|
mock_result.pages = [mock_page1, mock_page2, mock_page3]
|
|
|
|
mock_poller = Mock()
|
|
mock_poller.result.return_value = mock_result
|
|
mock_poller.details = {"operation_id": "test-456"}
|
|
|
|
mock_client = Mock()
|
|
mock_client.begin_analyze_document.return_value = mock_poller
|
|
mock_client.get_analyze_result_pdf.return_value = iter([b"content"])
|
|
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.document_intelligence_client",
|
|
mock_client,
|
|
), patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.process_with_azure_document_intelligence.rotate_pdf_pages"
|
|
) as mock_rotate:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_rotate.delay = MagicMock()
|
|
|
|
result = process_with_azure_document_intelligence.run(
|
|
filename=test_pdf.name, file_id=2
|
|
)
|
|
|
|
# Verify rotation data was passed correctly
|
|
mock_rotate.delay.assert_called_once()
|
|
call_args = mock_rotate.delay.call_args
|
|
rotation_data = call_args[0][2]
|
|
|
|
# Check rotation data structure (page indices as integers)
|
|
assert 0 in rotation_data and rotation_data[0] == 90
|
|
assert 1 not in rotation_data # Page 2 has no rotation
|
|
assert 2 in rotation_data and rotation_data[2] == 180
|
|
|
|
def test_azure_api_error_handling(self, tmp_path):
|
|
"""Test error handling when Azure API fails."""
|
|
# Create tmp directory and test PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
test_pdf.write_bytes(b"%PDF-1.4\n%%EOF")
|
|
|
|
mock_client = Mock()
|
|
mock_client.begin_analyze_document.side_effect = Exception(
|
|
"Azure API connection failed"
|
|
)
|
|
|
|
with patch(
|
|
"app.tasks.process_with_azure_document_intelligence.document_intelligence_client",
|
|
mock_client,
|
|
), patch(
|
|
"app.tasks.process_with_azure_document_intelligence.settings"
|
|
) as mock_settings:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
|
|
# Should raise the exception
|
|
with pytest.raises(Exception) as exc_info:
|
|
process_with_azure_document_intelligence.run(
|
|
filename=test_pdf.name, file_id=1
|
|
)
|
|
|
|
assert "Azure API connection failed" in str(exc_info.value)
|
|
|
|
def test_get_pdf_page_count_success(self, tmp_path):
|
|
"""Test successful page count extraction from PDF."""
|
|
# Create a PDF with text content
|
|
test_pdf = tmp_path / "test.pdf"
|
|
pdf_content = b"""%PDF-1.4
|
|
1 0 obj
|
|
<<
|
|
/Type /Catalog
|
|
/Pages 2 0 R
|
|
>>
|
|
endobj
|
|
2 0 obj
|
|
<<
|
|
/Type /Pages
|
|
/Kids [3 0 R 4 0 R]
|
|
/Count 2
|
|
>>
|
|
endobj
|
|
3 0 obj
|
|
<<
|
|
/Type /Page
|
|
/Parent 2 0 R
|
|
/MediaBox [0 0 612 792]
|
|
>>
|
|
endobj
|
|
4 0 obj
|
|
<<
|
|
/Type /Page
|
|
/Parent 2 0 R
|
|
/MediaBox [0 0 612 792]
|
|
>>
|
|
endobj
|
|
xref
|
|
0 5
|
|
0000000000 65535 f
|
|
0000000009 00000 n
|
|
0000000058 00000 n
|
|
0000000119 00000 n
|
|
0000000182 00000 n
|
|
trailer
|
|
<<
|
|
/Size 5
|
|
/Root 1 0 R
|
|
>>
|
|
startxref
|
|
245
|
|
%%EOF
|
|
"""
|
|
test_pdf.write_bytes(pdf_content)
|
|
|
|
page_count = get_pdf_page_count(str(test_pdf))
|
|
assert page_count == 2
|
|
|
|
def test_get_pdf_page_count_error(self, tmp_path):
|
|
"""Test page count returns None on error."""
|
|
# Create an invalid PDF file
|
|
test_pdf = tmp_path / "invalid.pdf"
|
|
test_pdf.write_bytes(b"not a valid pdf")
|
|
|
|
page_count = get_pdf_page_count(str(test_pdf))
|
|
assert page_count is None
|
|
|
|
def test_check_page_rotation_with_rotated_pages(self):
|
|
"""Test check_page_rotation with rotated pages."""
|
|
# Mock pages with rotation
|
|
mock_page1 = Mock()
|
|
mock_page1.angle = 90
|
|
|
|
mock_page2 = Mock()
|
|
mock_page2.angle = 0
|
|
|
|
mock_result = Mock()
|
|
mock_result.pages = [mock_page1, mock_page2]
|
|
|
|
rotation_data = check_page_rotation(mock_result, "test.pdf")
|
|
|
|
assert 0 in rotation_data
|
|
assert rotation_data[0] == 90
|
|
assert 1 not in rotation_data # No rotation for page 2
|
|
|
|
def test_check_page_rotation_no_pages(self):
|
|
"""Test check_page_rotation with no page information."""
|
|
mock_result = Mock()
|
|
mock_result.pages = None
|
|
|
|
rotation_data = check_page_rotation(mock_result, "test.pdf")
|
|
|
|
assert rotation_data == {}
|
|
|
|
def test_check_page_rotation_missing_angle_attribute(self):
|
|
"""Test check_page_rotation when pages don't have angle attribute."""
|
|
mock_page = Mock(spec=[]) # Page without 'angle' attribute
|
|
|
|
mock_result = Mock()
|
|
mock_result.pages = [mock_page]
|
|
|
|
rotation_data = check_page_rotation(mock_result, "test.pdf")
|
|
|
|
assert rotation_data == {}
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestRefineTextWithGPT:
|
|
"""Tests for OpenAI text refinement task."""
|
|
|
|
def test_successful_text_refinement(self):
|
|
"""Test successful text refinement with OpenAI."""
|
|
raw_text = "This is s0me text with OCR err0rs"
|
|
filename = "test.pdf"
|
|
|
|
# Mock OpenAI response
|
|
mock_choice = Mock()
|
|
mock_choice.message.content = "This is some text with OCR errors"
|
|
|
|
mock_response = Mock()
|
|
mock_response.choices = [mock_choice]
|
|
|
|
mock_client = Mock()
|
|
mock_client.chat.completions.create.return_value = mock_response
|
|
|
|
# Import the module to patch the correct function
|
|
from app.tasks import extract_metadata_with_gpt as metadata_module
|
|
|
|
with patch(
|
|
"app.tasks.refine_text_with_gpt.client", mock_client
|
|
), patch.object(
|
|
metadata_module, "extract_metadata_with_gpt"
|
|
) as mock_extract, patch(
|
|
"app.tasks.refine_text_with_gpt.settings"
|
|
) as mock_settings:
|
|
|
|
mock_settings.openai_model = "gpt-4"
|
|
mock_extract.delay = MagicMock()
|
|
|
|
result = refine_text_with_gpt.run(filename, raw_text)
|
|
|
|
# Verify results
|
|
assert result["filename"] == filename
|
|
assert result["cleaned_text"] == "This is some text with OCR errors"
|
|
|
|
# Verify OpenAI was called correctly
|
|
mock_client.chat.completions.create.assert_called_once()
|
|
call_kwargs = mock_client.chat.completions.create.call_args[1]
|
|
assert call_kwargs["model"] == "gpt-4"
|
|
assert len(call_kwargs["messages"]) == 2
|
|
assert call_kwargs["messages"][1]["content"] == raw_text
|
|
|
|
# Verify metadata extraction was queued
|
|
mock_extract.delay.assert_called_once_with(
|
|
filename, "This is some text with OCR errors"
|
|
)
|
|
|
|
def test_openai_api_error(self):
|
|
"""Test error handling when OpenAI API fails."""
|
|
raw_text = "Test text"
|
|
filename = "test.pdf"
|
|
|
|
mock_client = Mock()
|
|
mock_client.chat.completions.create.side_effect = Exception(
|
|
"OpenAI API error"
|
|
)
|
|
|
|
with patch(
|
|
"app.tasks.refine_text_with_gpt.client", mock_client
|
|
), patch(
|
|
"app.tasks.refine_text_with_gpt.settings"
|
|
) as mock_settings:
|
|
|
|
mock_settings.openai_model = "gpt-4"
|
|
|
|
# Should raise the exception
|
|
with pytest.raises(Exception) as exc_info:
|
|
refine_text_with_gpt.run(filename, raw_text)
|
|
|
|
assert "OpenAI API error" in str(exc_info.value)
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestRotatePdfPages:
|
|
"""Tests for PDF page rotation task."""
|
|
|
|
def test_determine_rotation_angle_exact_90(self):
|
|
"""Test rotation angle determination for exact 90 degrees."""
|
|
angle = determine_rotation_angle(90)
|
|
assert angle == 270 # PyPDF2 uses clockwise, so 360 - 90 = 270
|
|
|
|
def test_determine_rotation_angle_exact_180(self):
|
|
"""Test rotation angle determination for exact 180 degrees."""
|
|
angle = determine_rotation_angle(180)
|
|
assert angle == 180 # 360 - 180 = 180
|
|
|
|
def test_determine_rotation_angle_exact_270(self):
|
|
"""Test rotation angle determination for exact 270 degrees."""
|
|
angle = determine_rotation_angle(270)
|
|
assert angle == 90 # 360 - 270 = 90
|
|
|
|
def test_determine_rotation_angle_near_90(self):
|
|
"""Test rotation angle determination for angle near 90 degrees."""
|
|
angle = determine_rotation_angle(88) # Within 5° of 90
|
|
assert angle == 270
|
|
|
|
angle = determine_rotation_angle(93) # Within 5° of 90
|
|
assert angle == 270
|
|
|
|
def test_determine_rotation_angle_small_angle(self):
|
|
"""Test that very small angles are not rotated."""
|
|
angle = determine_rotation_angle(0.5)
|
|
assert angle == 0
|
|
|
|
angle = determine_rotation_angle(359.5)
|
|
assert angle == 0
|
|
|
|
def test_determine_rotation_angle_arbitrary(self):
|
|
"""Test rotation angle determination for arbitrary angles."""
|
|
# 45 degrees should round to nearest 90-degree increment (0 or 90)
|
|
angle = determine_rotation_angle(45)
|
|
# Round(45/90) * 90 = 0, so should return 0
|
|
assert angle == 0
|
|
|
|
# 135 degrees should round to 180
|
|
angle = determine_rotation_angle(135)
|
|
# Round(135/90) * 90 = 90, but it's closer to 180, so should return 180
|
|
assert angle == 180
|
|
|
|
def test_rotate_pdf_pages_with_rotation(self, tmp_path):
|
|
"""Test PDF rotation with rotation data."""
|
|
# Create tmp directory and test PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
pdf_content = b"""%PDF-1.4
|
|
1 0 obj
|
|
<<
|
|
/Type /Catalog
|
|
/Pages 2 0 R
|
|
>>
|
|
endobj
|
|
2 0 obj
|
|
<<
|
|
/Type /Pages
|
|
/Kids [3 0 R]
|
|
/Count 1
|
|
>>
|
|
endobj
|
|
3 0 obj
|
|
<<
|
|
/Type /Page
|
|
/Parent 2 0 R
|
|
/MediaBox [0 0 612 792]
|
|
>>
|
|
endobj
|
|
xref
|
|
0 4
|
|
0000000000 65535 f
|
|
0000000009 00000 n
|
|
0000000058 00000 n
|
|
0000000115 00000 n
|
|
trailer
|
|
<<
|
|
/Size 4
|
|
/Root 1 0 R
|
|
>>
|
|
startxref
|
|
197
|
|
%%EOF
|
|
"""
|
|
test_pdf.write_bytes(pdf_content)
|
|
|
|
rotation_data = {0: 90} # Rotate first page by 90 degrees
|
|
extracted_text = "Test text"
|
|
|
|
with patch(
|
|
"app.tasks.rotate_pdf_pages.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.rotate_pdf_pages.extract_metadata_with_gpt"
|
|
) as mock_extract:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_extract.delay = MagicMock()
|
|
|
|
result = rotate_pdf_pages.run(
|
|
filename=test_pdf.name,
|
|
extracted_text=extracted_text,
|
|
rotation_data=rotation_data,
|
|
file_id=1,
|
|
)
|
|
|
|
# Verify results
|
|
assert result["file"] == test_pdf.name
|
|
assert result["status"] == "rotated"
|
|
assert "applied_rotations" in result
|
|
assert "0" in result["applied_rotations"]
|
|
|
|
# Verify metadata extraction was queued
|
|
mock_extract.delay.assert_called_once_with(
|
|
test_pdf.name, extracted_text, 1
|
|
)
|
|
|
|
def test_rotate_pdf_pages_no_rotation_needed(self, tmp_path):
|
|
"""Test PDF rotation when no rotation is needed."""
|
|
# Create tmp directory and test PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
test_pdf.write_bytes(b"%PDF-1.4\n%%EOF")
|
|
|
|
extracted_text = "Test text"
|
|
|
|
with patch(
|
|
"app.tasks.rotate_pdf_pages.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.rotate_pdf_pages.extract_metadata_with_gpt"
|
|
) as mock_extract:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_extract.delay = MagicMock()
|
|
|
|
# Call with no rotation data
|
|
result = rotate_pdf_pages.run(
|
|
filename=test_pdf.name,
|
|
extracted_text=extracted_text,
|
|
rotation_data=None,
|
|
file_id=1,
|
|
)
|
|
|
|
# Verify results
|
|
assert result["status"] == "no_rotation_needed"
|
|
|
|
# Verify metadata extraction was still queued
|
|
mock_extract.delay.assert_called_once()
|
|
|
|
def test_rotate_pdf_pages_zero_rotations(self, tmp_path):
|
|
"""Test PDF rotation when rotation data contains only zero angles."""
|
|
# Create tmp directory and test PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
test_pdf.write_bytes(b"%PDF-1.4\n%%EOF")
|
|
|
|
rotation_data = {0: 0, 1: 0} # All pages have 0 rotation
|
|
extracted_text = "Test text"
|
|
|
|
with patch(
|
|
"app.tasks.rotate_pdf_pages.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.rotate_pdf_pages.extract_metadata_with_gpt"
|
|
) as mock_extract:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_extract.delay = MagicMock()
|
|
|
|
result = rotate_pdf_pages.run(
|
|
filename=test_pdf.name,
|
|
extracted_text=extracted_text,
|
|
rotation_data=rotation_data,
|
|
file_id=1,
|
|
)
|
|
|
|
# Verify no rotation was applied
|
|
assert result["status"] == "no_rotation_needed"
|
|
|
|
def test_rotate_pdf_pages_file_not_found(self, tmp_path):
|
|
"""Test error handling when PDF file is not found."""
|
|
# Create tmp directory but no PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
|
|
with patch(
|
|
"app.tasks.rotate_pdf_pages.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.rotate_pdf_pages.extract_metadata_with_gpt"
|
|
) as mock_extract:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_extract.delay = MagicMock()
|
|
|
|
# The function catches FileNotFoundError and returns error status
|
|
result = rotate_pdf_pages.run(
|
|
filename="nonexistent.pdf",
|
|
extracted_text="text",
|
|
rotation_data={0: 90},
|
|
file_id=1,
|
|
)
|
|
|
|
# Verify error handling
|
|
assert result["status"] == "rotation_failed"
|
|
assert "error" in result
|
|
assert "PDF file not found" in result["error"]
|
|
|
|
# Verify metadata extraction was still queued
|
|
mock_extract.delay.assert_called_once()
|
|
|
|
def test_rotate_pdf_pages_continues_on_error(self, tmp_path):
|
|
"""Test that metadata extraction continues even if rotation fails."""
|
|
# Create tmp directory and invalid PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
test_pdf.write_bytes(b"invalid pdf")
|
|
|
|
rotation_data = {0: 90}
|
|
extracted_text = "Test text"
|
|
|
|
with patch(
|
|
"app.tasks.rotate_pdf_pages.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.rotate_pdf_pages.extract_metadata_with_gpt"
|
|
) as mock_extract:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_extract.delay = MagicMock()
|
|
|
|
result = rotate_pdf_pages.run(
|
|
filename=test_pdf.name,
|
|
extracted_text=extracted_text,
|
|
rotation_data=rotation_data,
|
|
file_id=1,
|
|
)
|
|
|
|
# Verify error was handled
|
|
assert result["status"] == "rotation_failed"
|
|
assert "error" in result
|
|
|
|
# Verify metadata extraction was still queued
|
|
mock_extract.delay.assert_called_once()
|
|
|
|
def test_rotate_pdf_pages_with_string_keys(self, tmp_path):
|
|
"""Test that rotation data with string keys is properly handled."""
|
|
# Create tmp directory and test PDF
|
|
tmp_dir = tmp_path / "tmp"
|
|
tmp_dir.mkdir()
|
|
test_pdf = tmp_dir / "test.pdf"
|
|
pdf_content = b"""%PDF-1.4
|
|
1 0 obj
|
|
<<
|
|
/Type /Catalog
|
|
/Pages 2 0 R
|
|
>>
|
|
endobj
|
|
2 0 obj
|
|
<<
|
|
/Type /Pages
|
|
/Kids [3 0 R]
|
|
/Count 1
|
|
>>
|
|
endobj
|
|
3 0 obj
|
|
<<
|
|
/Type /Page
|
|
/Parent 2 0 R
|
|
/MediaBox [0 0 612 792]
|
|
>>
|
|
endobj
|
|
xref
|
|
0 4
|
|
0000000000 65535 f
|
|
0000000009 00000 n
|
|
0000000058 00000 n
|
|
0000000115 00000 n
|
|
trailer
|
|
<<
|
|
/Size 4
|
|
/Root 1 0 R
|
|
>>
|
|
startxref
|
|
197
|
|
%%EOF
|
|
"""
|
|
test_pdf.write_bytes(pdf_content)
|
|
|
|
# Use string keys instead of integer keys
|
|
rotation_data = {"0": "90"}
|
|
extracted_text = "Test text"
|
|
|
|
with patch(
|
|
"app.tasks.rotate_pdf_pages.settings"
|
|
) as mock_settings, patch(
|
|
"app.tasks.rotate_pdf_pages.extract_metadata_with_gpt"
|
|
) as mock_extract:
|
|
|
|
mock_settings.workdir = str(tmp_path)
|
|
mock_extract.delay = MagicMock()
|
|
|
|
result = rotate_pdf_pages.run(
|
|
filename=test_pdf.name,
|
|
extracted_text=extracted_text,
|
|
rotation_data=rotation_data,
|
|
file_id=1,
|
|
)
|
|
|
|
# Verify rotation was applied despite string keys
|
|
assert result["status"] == "rotated"
|
|
assert "applied_rotations" in result
|