fix: merge main branch and renumber migration 027→037
Resolve 3 merge conflicts and renumber the automation_hooks migration to follow main's migration chain (036_add_document_translation_fields). Conflicts resolved: - app/api/__init__.py: add automation_router alongside main's new routers - app/utils/settings_service.py: add automation_hooks_enabled alongside compliance_enabled - tests/conftest.py: add AutomationHook alongside AuditLog/ComplianceTemplate imports Migration renumbered: - 027_add_automation_hooks → 037_add_automation_hooks - down_revision: 026_add_scheduled_jobs → 036_add_document_translation_fields Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
@@ -121,6 +121,95 @@ class TestExtractMetadataFromFile:
|
||||
|
||||
assert result == {}
|
||||
|
||||
def test_extract_metadata_from_pdf(self, tmp_path):
|
||||
"""Test extracting metadata from a PDF file using pypdf when JSON is missing."""
|
||||
import pypdf
|
||||
|
||||
file_path = tmp_path / "test.pdf"
|
||||
|
||||
# Create a test PDF with metadata
|
||||
writer = pypdf.PdfWriter()
|
||||
writer.add_blank_page(width=100, height=100)
|
||||
writer.add_metadata(
|
||||
{
|
||||
"/Title": "Test Title",
|
||||
"/Author": "Test Author",
|
||||
"/Subject": "Test Document",
|
||||
"/Keywords": "test, metadata, pypdf",
|
||||
}
|
||||
)
|
||||
with open(file_path, "wb") as f:
|
||||
writer.write(f)
|
||||
|
||||
result = extract_metadata_from_file(str(file_path))
|
||||
|
||||
# Keys are mapped to application-specific names
|
||||
assert result.get("filename") == "Test Title"
|
||||
assert result.get("absender") == "Test Author"
|
||||
assert result.get("document_type") == "Test Document"
|
||||
assert result.get("tags") == "test, metadata, pypdf"
|
||||
|
||||
def test_extracts_embedded_metadata_from_pdf(self, tmp_path):
|
||||
"""Test that embedded PDF metadata is mapped to application-specific keys."""
|
||||
import pypdf
|
||||
|
||||
file_path = tmp_path / "mapped.pdf"
|
||||
|
||||
writer = pypdf.PdfWriter()
|
||||
writer.add_blank_page(width=100, height=100)
|
||||
writer.add_metadata(
|
||||
{
|
||||
"/Title": "Invoice 2024",
|
||||
"/Author": "Acme Corp",
|
||||
"/Subject": "invoice",
|
||||
"/Keywords": "finance, billing",
|
||||
}
|
||||
)
|
||||
with open(file_path, "wb") as f:
|
||||
writer.write(f)
|
||||
|
||||
result = extract_metadata_from_file(str(file_path))
|
||||
|
||||
# Verify the PDF-to-app key mapping
|
||||
assert result["filename"] == "Invoice 2024"
|
||||
assert result["absender"] == "Acme Corp"
|
||||
assert result["document_type"] == "invoice"
|
||||
assert result["tags"] == "finance, billing"
|
||||
|
||||
def test_pdf_metadata_does_not_overwrite_json(self, tmp_path):
|
||||
"""Test that JSON metadata takes precedence over embedded PDF metadata."""
|
||||
import pypdf
|
||||
|
||||
file_path = tmp_path / "dual.pdf"
|
||||
|
||||
# Create a PDF with embedded metadata
|
||||
writer = pypdf.PdfWriter()
|
||||
writer.add_blank_page(width=100, height=100)
|
||||
writer.add_metadata(
|
||||
{
|
||||
"/Title": "PDF Title",
|
||||
"/Author": "PDF Author",
|
||||
"/Subject": "PDF Subject",
|
||||
"/Keywords": "pdf, keywords",
|
||||
}
|
||||
)
|
||||
with open(file_path, "wb") as f:
|
||||
writer.write(f)
|
||||
|
||||
# Create a companion JSON file that sets some overlapping fields
|
||||
json_metadata = {"filename": "JSON Filename", "absender": "JSON Author"}
|
||||
json_path = tmp_path / "dual.json"
|
||||
json_path.write_text(json.dumps(json_metadata))
|
||||
|
||||
result = extract_metadata_from_file(str(file_path))
|
||||
|
||||
# JSON values must not be overwritten by PDF metadata
|
||||
assert result["filename"] == "JSON Filename"
|
||||
assert result["absender"] == "JSON Author"
|
||||
# Fields missing from JSON are filled from PDF metadata
|
||||
assert result["document_type"] == "PDF Subject"
|
||||
assert result["tags"] == "pdf, keywords"
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestAttachLogo:
|
||||
|
||||
Reference in New Issue
Block a user