From ef484f85a412549107e1e4bd640bb5488b757ce5 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Mon, 9 Feb 2026 20:53:50 +0000 Subject: [PATCH 1/4] Initial plan From 2bcd774d6d1cfcd3b90babbc447c61b33e384549 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Mon, 9 Feb 2026 21:00:51 +0000 Subject: [PATCH 2/4] fix(security): enhance path traversal protection in file uploads - Import and use sanitize_filename utility in ui_upload endpoint - Enhance sanitize_filename to handle Windows-style paths (backslashes) - Add protection against path traversal patterns (..) - Replace all path separators with underscores - Add comprehensive security tests for Windows-style paths and mixed separators - All existing tests pass with improved security This addresses the "Uncontrolled data used in path expression" code scanning alert by ensuring all user-provided filenames are properly sanitized before being used in any file operations or stored in the database. Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com> --- app/api/files.py | 6 ++++- app/utils/filename_utils.py | 22 +++++++++++++---- tests/test_file_upload.py | 48 +++++++++++++++++++++++++++++++++++-- 3 files changed, 69 insertions(+), 7 deletions(-) diff --git a/app/api/files.py b/app/api/files.py index 58e0d834..e90e8d3d 100644 --- a/app/api/files.py +++ b/app/api/files.py @@ -19,6 +19,7 @@ from app.models import FileRecord, ProcessingLog from app.tasks.convert_to_pdf import convert_to_pdf from app.tasks.process_document import process_document from app.utils.file_status import get_files_processing_status +from app.utils.filename_utils import sanitize_filename # Set up logging logger = logging.getLogger(__name__) @@ -657,7 +658,10 @@ async def ui_upload(request: Request, file: UploadFile = File(...)): workdir = settings.workdir # Extract just the filename without any path components to prevent path traversal - safe_filename = os.path.basename(file.filename) + # First, use basename to remove any directory components + base_filename = os.path.basename(file.filename) + # Then sanitize the filename to remove special characters and ensure filesystem compatibility + safe_filename = sanitize_filename(base_filename) # Generate a unique filename with UUID to prevent overwriting and filename conflicts unique_id = str(uuid.uuid4()) diff --git a/app/utils/filename_utils.py b/app/utils/filename_utils.py index b4cf5ad5..a491613a 100644 --- a/app/utils/filename_utils.py +++ b/app/utils/filename_utils.py @@ -71,24 +71,38 @@ def get_unique_filename(original_path, check_exists_func=None): def sanitize_filename(filename): """ - Sanitize a filename to ensure it's valid across different file systems. + Sanitize a filename to ensure it's valid across different file systems + and prevent path traversal attacks. Args: filename (str): The filename to sanitize Returns: str: A sanitized filename + + Security: + - Removes path separators (/ and \) + - Prevents path traversal patterns (..) + - Replaces problematic characters with underscores + - Ensures compatibility across Windows, Linux, and macOS """ + # First, replace all path separators (both Unix and Windows style) with underscores + sanitized = filename.replace("/", "_").replace("\\", "_") + # Replace characters that are problematic in various filesystems # Keep only alphanumeric, dash, underscore, period, and space - sanitized = re.sub(r"[^\w\-\. ]", "_", filename) + sanitized = re.sub(r"[^\w\-\. ]", "_", sanitized) + + # Remove or replace path traversal patterns + # Replace consecutive dots with a single underscore to prevent .. patterns + sanitized = re.sub(r"\.\.+", "_", sanitized) # Replace multiple spaces/underscores with single ones sanitized = re.sub(r"__+", "_", sanitized) sanitized = re.sub(r" +", " ", sanitized) - # Trim leading/trailing spaces and periods which cause issues in Windows - sanitized = sanitized.strip(". ") + # Trim leading/trailing spaces, periods, and underscores which cause issues in Windows + sanitized = sanitized.strip(". _") # Ensure the filename isn't empty after sanitization if not sanitized or sanitized == ".": diff --git a/tests/test_file_upload.py b/tests/test_file_upload.py index 07243a47..24feacb1 100644 --- a/tests/test_file_upload.py +++ b/tests/test_file_upload.py @@ -272,10 +272,54 @@ class TestUploadSecurity: assert response.status_code == 200 data = response.json() - # Original filename should be preserved (sanitized by basename) - assert data["original_filename"] == special_filename + # Original filename should be sanitized (special chars replaced with underscores) + assert "!" not in data["original_filename"] + assert "@" not in data["original_filename"] + assert "#" not in data["original_filename"] + # Spaces and basic characters should be preserved + assert "file" in data["original_filename"] + assert ".pdf" in data["original_filename"] # Stored filename should have UUID and extension assert data["stored_filename"].endswith(".pdf") + + def test_path_traversal_prevention_windows_style(self, client: TestClient, mock_celery_tasks): + """Test that Windows-style path traversal attempts are prevented.""" + # Try Windows-style path with backslashes + malicious_filename = "..\\..\\..\\windows\\system32\\config.pdf" + pdf_content = b"%PDF-1.4\n%EOF" + + response = client.post( + "/api/ui-upload", + files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} + ) + + assert response.status_code == 200 + data = response.json() + + # Backslashes should be removed or replaced + assert "\\" not in data["original_filename"] + assert ".." not in data["original_filename"] + # Should contain sanitized version + assert "config.pdf" in data["original_filename"] + + def test_path_traversal_prevention_mixed_separators(self, client: TestClient, mock_celery_tasks): + """Test handling of filenames with mixed path separators.""" + malicious_filename = "../path\\to/file.pdf" + pdf_content = b"%PDF-1.4\n%EOF" + + response = client.post( + "/api/ui-upload", + files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} + ) + + assert response.status_code == 200 + data = response.json() + + # Should only keep the filename without any path components + assert "/" not in data["original_filename"] + assert "\\" not in data["original_filename"] + assert ".." not in data["original_filename"] + assert "file.pdf" in data["original_filename"] @pytest.mark.integration From 43b512fee8f4964b6ad204c45188ae1da9564c48 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Mon, 9 Feb 2026 21:04:03 +0000 Subject: [PATCH 3/4] refactor(security): improve sanitization logic and tests based on code review - Change consecutive dots regex to simple replace for better precision - Update tests to verify exact sanitized output - Fix docstring syntax warning with raw string - Add detailed comments explaining sanitization behavior - All 43 tests pass (21 file upload + 22 filename utils) Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com> --- app/utils/filename_utils.py | 6 +++--- tests/test_file_upload.py | 20 +++++++++++++++----- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/app/utils/filename_utils.py b/app/utils/filename_utils.py index a491613a..92751743 100644 --- a/app/utils/filename_utils.py +++ b/app/utils/filename_utils.py @@ -70,7 +70,7 @@ def get_unique_filename(original_path, check_exists_func=None): def sanitize_filename(filename): - """ + r""" Sanitize a filename to ensure it's valid across different file systems and prevent path traversal attacks. @@ -94,8 +94,8 @@ def sanitize_filename(filename): sanitized = re.sub(r"[^\w\-\. ]", "_", sanitized) # Remove or replace path traversal patterns - # Replace consecutive dots with a single underscore to prevent .. patterns - sanitized = re.sub(r"\.\.+", "_", sanitized) + # Replace specifically '..' to prevent path traversal while preserving single dots + sanitized = sanitized.replace("..", "_") # Replace multiple spaces/underscores with single ones sanitized = re.sub(r"__+", "_", sanitized) diff --git a/tests/test_file_upload.py b/tests/test_file_upload.py index 24feacb1..5687df29 100644 --- a/tests/test_file_upload.py +++ b/tests/test_file_upload.py @@ -296,11 +296,18 @@ class TestUploadSecurity: assert response.status_code == 200 data = response.json() - # Backslashes should be removed or replaced + # On Unix systems, os.path.basename doesn't recognize Windows backslashes, + # so the full string goes to sanitize_filename which then: + # 1. Replaces backslashes with underscores + # 2. Replaces .. with underscores + # 3. Strips leading/trailing underscores + # Result: "windows_system32_config.pdf" + expected_filename = "windows_system32_config.pdf" + assert data["original_filename"] == expected_filename + # Verify no dangerous characters remain assert "\\" not in data["original_filename"] assert ".." not in data["original_filename"] - # Should contain sanitized version - assert "config.pdf" in data["original_filename"] + assert "/" not in data["original_filename"] def test_path_traversal_prevention_mixed_separators(self, client: TestClient, mock_celery_tasks): """Test handling of filenames with mixed path separators.""" @@ -315,11 +322,14 @@ class TestUploadSecurity: assert response.status_code == 200 data = response.json() - # Should only keep the filename without any path components + # os.path.basename recognizes Unix / separators and extracts "file.pdf" + # sanitize_filename then ensures it's safe (already is in this case) + expected_filename = "file.pdf" + assert data["original_filename"] == expected_filename + # Verify no dangerous characters remain assert "/" not in data["original_filename"] assert "\\" not in data["original_filename"] assert ".." not in data["original_filename"] - assert "file.pdf" in data["original_filename"] @pytest.mark.integration From 3d2b21b3c8448dc11ce78c44a38579332ce910d9 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Mon, 9 Feb 2026 21:05:44 +0000 Subject: [PATCH 4/4] style: fix linting issues (whitespace and formatting) - Remove trailing whitespace from blank lines - Apply black formatting to test file - All tests still pass Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com> --- app/utils/filename_utils.py | 4 +- tests/test_file_upload.py | 252 ++++++++++++++++-------------------- 2 files changed, 113 insertions(+), 143 deletions(-) diff --git a/app/utils/filename_utils.py b/app/utils/filename_utils.py index 92751743..83b8baff 100644 --- a/app/utils/filename_utils.py +++ b/app/utils/filename_utils.py @@ -88,11 +88,11 @@ def sanitize_filename(filename): """ # First, replace all path separators (both Unix and Windows style) with underscores sanitized = filename.replace("/", "_").replace("\\", "_") - + # Replace characters that are problematic in various filesystems # Keep only alphanumeric, dash, underscore, period, and space sanitized = re.sub(r"[^\w\-\. ]", "_", sanitized) - + # Remove or replace path traversal patterns # Replace specifically '..' to prevent path traversal while preserving single dots sanitized = sanitized.replace("..", "_") diff --git a/tests/test_file_upload.py b/tests/test_file_upload.py index 5687df29..7ffd2a93 100644 --- a/tests/test_file_upload.py +++ b/tests/test_file_upload.py @@ -9,6 +9,7 @@ Tests cover: - Security issues (path traversal) - Error handling """ + import io import os import pytest @@ -21,68 +22,63 @@ from fastapi.testclient import TestClient def mock_celery_tasks(): """Mock all Celery tasks to prevent execution.""" # Patch the entire task object where it's used (in app.api.files) - with patch("app.api.files.process_document") as mock_process_task, \ - patch("app.api.files.convert_to_pdf") as mock_convert_task: - + with ( + patch("app.api.files.process_document") as mock_process_task, + patch("app.api.files.convert_to_pdf") as mock_convert_task, + ): + # Setup default return values for .delay() mock_task = MagicMock() mock_task.id = "test-task-id-123" mock_process_task.delay.return_value = mock_task mock_convert_task.delay.return_value = mock_task - - yield { - "process_document": mock_process_task.delay, - "convert_to_pdf": mock_convert_task.delay - } + + yield {"process_document": mock_process_task.delay, "convert_to_pdf": mock_convert_task.delay} @pytest.mark.integration class TestValidFileUploads: """Tests for successful file uploads with various valid file types.""" - + def test_upload_valid_pdf(self, client: TestClient, sample_pdf_path: str, mock_celery_tasks): """Test uploading a valid PDF file.""" with open(sample_pdf_path, "rb") as f: - response = client.post( - "/api/ui-upload", - files={"file": ("document.pdf", f, "application/pdf")} - ) - + response = client.post("/api/ui-upload", files={"file": ("document.pdf", f, "application/pdf")}) + assert response.status_code == 200 data = response.json() - + # Verify response structure assert "task_id" in data assert "status" in data assert "original_filename" in data assert "stored_filename" in data - + # Verify response values assert data["status"] == "queued" assert data["original_filename"] == "document.pdf" assert data["stored_filename"].endswith(".pdf") - + # Verify the processing task was called mock_celery_tasks["process_document"].assert_called_once() call_args = mock_celery_tasks["process_document"].call_args assert call_args.kwargs["original_filename"] == "document.pdf" - + def test_upload_valid_text_file(self, client: TestClient, mock_celery_tasks): """Test uploading a valid text file.""" text_content = b"This is a test text file.\nWith multiple lines." response = client.post( - "/api/ui-upload", - files={"file": ("document.txt", io.BytesIO(text_content), "text/plain")} + "/api/ui-upload", files={"file": ("document.txt", io.BytesIO(text_content), "text/plain")} ) - + assert response.status_code == 200 data = response.json() assert data["original_filename"] == "document.txt" assert data["stored_filename"].endswith(".txt") - + # Text files should be converted to PDF mock_celery_tasks["convert_to_pdf"].assert_called_once() - + def test_upload_valid_image_jpeg(self, client: TestClient, mock_celery_tasks): """Test uploading a valid JPEG image.""" # Create a minimal valid JPEG @@ -91,19 +87,16 @@ class TestValidFileUploads: b"\xff\xdb\x00C\x00\x08\x06\x06\x07\x06\x05\x08\x07\x07\x07\t\t\x08\n" b"\xff\xd9" ) - - response = client.post( - "/api/ui-upload", - files={"file": ("image.jpg", io.BytesIO(jpeg_content), "image/jpeg")} - ) - + + response = client.post("/api/ui-upload", files={"file": ("image.jpg", io.BytesIO(jpeg_content), "image/jpeg")}) + assert response.status_code == 200 data = response.json() assert data["original_filename"] == "image.jpg" - + # Images should be converted to PDF mock_celery_tasks["convert_to_pdf"].assert_called_once() - + def test_upload_valid_png_image(self, client: TestClient, mock_celery_tasks): """Test uploading a valid PNG image.""" # Create a minimal valid PNG (1x1 transparent pixel) @@ -112,51 +105,47 @@ class TestValidFileUploads: b"\x08\x06\x00\x00\x00\x1f\x15\xc4\x89\x00\x00\x00\nIDATx\x9cc\x00\x01" b"\x00\x00\x05\x00\x01\r\n-\xb4\x00\x00\x00\x00IEND\xaeB`\x82" ) - + response = client.post( - "/api/ui-upload", - files={"file": ("screenshot.png", io.BytesIO(png_content), "image/png")} + "/api/ui-upload", files={"file": ("screenshot.png", io.BytesIO(png_content), "image/png")} ) - + assert response.status_code == 200 data = response.json() assert data["original_filename"] == "screenshot.png" assert data["stored_filename"].endswith(".png") mock_celery_tasks["convert_to_pdf"].assert_called_once() - + def test_upload_office_document_docx(self, client: TestClient, mock_celery_tasks): """Test uploading a Word document.""" # Create minimal DOCX content (ZIP file with proper structure) - docx_content = ( - b"PK\x03\x04\x14\x00\x00\x00\x08\x00" + b"\x00" * 50 - ) - + docx_content = b"PK\x03\x04\x14\x00\x00\x00\x08\x00" + b"\x00" * 50 + response = client.post( "/api/ui-upload", - files={"file": ( - "report.docx", - io.BytesIO(docx_content), - "application/vnd.openxmlformats-officedocument.wordprocessingml.document" - )} + files={ + "file": ( + "report.docx", + io.BytesIO(docx_content), + "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + ) + }, ) - + assert response.status_code == 200 data = response.json() assert data["original_filename"] == "report.docx" assert data["stored_filename"].endswith(".docx") - + # Office documents should be converted to PDF mock_celery_tasks["convert_to_pdf"].assert_called_once() - + def test_upload_csv_file(self, client: TestClient, mock_celery_tasks): """Test uploading a CSV file.""" csv_content = b"name,age,city\nJohn,30,NYC\nJane,25,LA\n" - - response = client.post( - "/api/ui-upload", - files={"file": ("data.csv", io.BytesIO(csv_content), "text/csv")} - ) - + + response = client.post("/api/ui-upload", files={"file": ("data.csv", io.BytesIO(csv_content), "text/csv")}) + assert response.status_code == 200 data = response.json() assert data["original_filename"] == "data.csv" @@ -166,54 +155,49 @@ class TestValidFileUploads: @pytest.mark.integration class TestInvalidFileUploads: """Tests for handling invalid or problematic file uploads.""" - + def test_upload_file_too_large(self, client: TestClient): """Test that files over 500MB are rejected.""" # Create a large file content (mock it to avoid memory issues) large_content = b"x" * 1024 # 1KB for testing - + with patch("os.path.getsize") as mock_getsize: # Mock the file size to be over 500MB mock_getsize.return_value = 501 * 1024 * 1024 # 501MB - + response = client.post( - "/api/ui-upload", - files={"file": ("huge.pdf", io.BytesIO(large_content), "application/pdf")} + "/api/ui-upload", files={"file": ("huge.pdf", io.BytesIO(large_content), "application/pdf")} ) - + assert response.status_code == 413 # Request Entity Too Large assert "too large" in response.json()["detail"].lower() - + def test_upload_executable_file(self, client: TestClient, mock_celery_tasks): """Test that executable files are handled (attempted conversion).""" exe_content = b"MZ\x90\x00" # PE header - + response = client.post( - "/api/ui-upload", - files={"file": ("program.exe", io.BytesIO(exe_content), "application/x-msdownload")} + "/api/ui-upload", files={"file": ("program.exe", io.BytesIO(exe_content), "application/x-msdownload")} ) - + # Per the code, unsupported types get a warning but are still processed assert response.status_code == 200 # The system attempts conversion even for unsupported types mock_celery_tasks["convert_to_pdf"].assert_called_once() - + def test_upload_empty_file(self, client: TestClient, mock_celery_tasks): """Test uploading an empty file.""" - response = client.post( - "/api/ui-upload", - files={"file": ("empty.txt", io.BytesIO(b""), "text/plain")} - ) - + response = client.post("/api/ui-upload", files={"file": ("empty.txt", io.BytesIO(b""), "text/plain")}) + # Empty files are accepted and queued for processing assert response.status_code == 200 data = response.json() assert data["original_filename"] == "empty.txt" - + def test_upload_without_file(self, client: TestClient): """Test POST request without a file.""" response = client.post("/api/ui-upload") - + # FastAPI should return a validation error assert response.status_code == 422 # Unprocessable Entity @@ -222,56 +206,53 @@ class TestInvalidFileUploads: @pytest.mark.security class TestUploadSecurity: """Tests for security aspects of file uploads.""" - + def test_path_traversal_prevention_dotdot(self, client: TestClient, mock_celery_tasks): """Test that path traversal attempts are prevented.""" # Try to upload a file with path traversal in filename malicious_filename = "../../etc/passwd.pdf" pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} ) - + assert response.status_code == 200 data = response.json() - + # The original filename should be sanitized (only basename) assert data["original_filename"] == "passwd.pdf" assert ".." not in data["stored_filename"] assert "/" not in data["stored_filename"] - + def test_path_traversal_prevention_absolute(self, client: TestClient, mock_celery_tasks): """Test that absolute path attempts are prevented.""" malicious_filename = "/etc/shadow.pdf" pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} ) - + assert response.status_code == 200 data = response.json() - + # Only the basename should be kept assert data["original_filename"] == "shadow.pdf" assert not data["stored_filename"].startswith("/") - + def test_filename_with_special_characters(self, client: TestClient, mock_celery_tasks): """Test handling of filenames with special characters.""" special_filename = "file name with spaces & special!@#chars.pdf" pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": (special_filename, io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": (special_filename, io.BytesIO(pdf_content), "application/pdf")} ) - + assert response.status_code == 200 data = response.json() - + # Original filename should be sanitized (special chars replaced with underscores) assert "!" not in data["original_filename"] assert "@" not in data["original_filename"] @@ -281,21 +262,20 @@ class TestUploadSecurity: assert ".pdf" in data["original_filename"] # Stored filename should have UUID and extension assert data["stored_filename"].endswith(".pdf") - + def test_path_traversal_prevention_windows_style(self, client: TestClient, mock_celery_tasks): """Test that Windows-style path traversal attempts are prevented.""" # Try Windows-style path with backslashes malicious_filename = "..\\..\\..\\windows\\system32\\config.pdf" pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} ) - + assert response.status_code == 200 data = response.json() - + # On Unix systems, os.path.basename doesn't recognize Windows backslashes, # so the full string goes to sanitize_filename which then: # 1. Replaces backslashes with underscores @@ -308,20 +288,19 @@ class TestUploadSecurity: assert "\\" not in data["original_filename"] assert ".." not in data["original_filename"] assert "/" not in data["original_filename"] - + def test_path_traversal_prevention_mixed_separators(self, client: TestClient, mock_celery_tasks): """Test handling of filenames with mixed path separators.""" malicious_filename = "../path\\to/file.pdf" pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": (malicious_filename, io.BytesIO(pdf_content), "application/pdf")} ) - + assert response.status_code == 200 data = response.json() - + # os.path.basename recognizes Unix / separators and extracts "file.pdf" # sanitize_filename then ensures it's safe (already is in this case) expected_filename = "file.pdf" @@ -335,75 +314,68 @@ class TestUploadSecurity: @pytest.mark.integration class TestUploadErrorHandling: """Tests for error handling during file uploads.""" - + def test_upload_disk_write_failure(self, client: TestClient): """Test handling of disk write failures.""" with patch("builtins.open", side_effect=IOError("Disk full")): pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": ("test.pdf", io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": ("test.pdf", io.BytesIO(pdf_content), "application/pdf")} ) - + assert response.status_code == 500 assert "Failed to save file" in response.json()["detail"] - + def test_upload_celery_task_failure(self, client: TestClient, mock_celery_tasks): """Test handling when Celery task queueing fails.""" mock_celery_tasks["process_document"].side_effect = Exception("Celery connection failed") - + pdf_content = b"%PDF-1.4\n%EOF" - + # The endpoint should still handle the error gracefully # In this case, the exception will propagate with pytest.raises(Exception): - client.post( - "/api/ui-upload", - files={"file": ("test.pdf", io.BytesIO(pdf_content), "application/pdf")} - ) + client.post("/api/ui-upload", files={"file": ("test.pdf", io.BytesIO(pdf_content), "application/pdf")}) @pytest.mark.integration class TestUploadFilenameHandling: """Tests for filename handling and UUID generation.""" - + def test_unique_filename_generation(self, client: TestClient, mock_celery_tasks): """Test that uploaded files get unique UUIDs.""" pdf_content = b"%PDF-1.4\n%EOF" - + # Upload same file twice response1 = client.post( - "/api/ui-upload", - files={"file": ("same.pdf", io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": ("same.pdf", io.BytesIO(pdf_content), "application/pdf")} ) - + response2 = client.post( - "/api/ui-upload", - files={"file": ("same.pdf", io.BytesIO(pdf_content), "application/pdf")} + "/api/ui-upload", files={"file": ("same.pdf", io.BytesIO(pdf_content), "application/pdf")} ) - + assert response1.status_code == 200 assert response2.status_code == 200 - + data1 = response1.json() data2 = response2.json() - + # Original filenames should be the same assert data1["original_filename"] == data2["original_filename"] == "same.pdf" - + # But stored filenames should be different (unique UUIDs) assert data1["stored_filename"] != data2["stored_filename"] - + def test_filename_without_extension(self, client: TestClient, mock_celery_tasks): """Test handling of files without extensions.""" content = b"Some content" - + response = client.post( - "/api/ui-upload", - files={"file": ("NOEXTENSION", io.BytesIO(content), "application/octet-stream")} + "/api/ui-upload", files={"file": ("NOEXTENSION", io.BytesIO(content), "application/octet-stream")} ) - + assert response.status_code == 200 data = response.json() assert data["original_filename"] == "NOEXTENSION" @@ -414,30 +386,28 @@ class TestUploadFilenameHandling: @pytest.mark.integration class TestUploadMimeTypeDetection: """Tests for MIME type detection and routing.""" - + def test_pdf_by_extension_only(self, client: TestClient, mock_celery_tasks): """Test PDF detection by file extension when MIME type is generic.""" pdf_content = b"%PDF-1.4\n%EOF" - + response = client.post( - "/api/ui-upload", - files={"file": ("doc.pdf", io.BytesIO(pdf_content), "application/octet-stream")} + "/api/ui-upload", files={"file": ("doc.pdf", io.BytesIO(pdf_content), "application/octet-stream")} ) - + assert response.status_code == 200 # Should route to process_document (not convert_to_pdf) based on extension mock_celery_tasks["process_document"].assert_called_once() - + def test_image_by_extension(self, client: TestClient, mock_celery_tasks): """Test image detection by file extension.""" # Generic binary content with image extension image_content = b"\x00\x01\x02\x03" - + response = client.post( - "/api/ui-upload", - files={"file": ("photo.jpg", io.BytesIO(image_content), "application/octet-stream")} + "/api/ui-upload", files={"file": ("photo.jpg", io.BytesIO(image_content), "application/octet-stream")} ) - + assert response.status_code == 200 # Should route to convert_to_pdf based on .jpg extension mock_celery_tasks["convert_to_pdf"].assert_called_once()