Merge pull request #119 from christianlouis/copilot/improve-file-detail-view

Add file reprocessing, process flow visualization, and PDF preview to detail view
This commit is contained in:
Christian Krakau-Louis
2026-02-07 22:40:47 +01:00
committed by GitHub
7 changed files with 635 additions and 6 deletions
+1 -1
View File
@@ -1 +1 @@
0.4.5-dev
0.1.0-test
+116
View File
@@ -363,6 +363,122 @@ def bulk_reprocess_files(request: Request, file_ids: List[int], db: Session = De
raise HTTPException(status_code=500, detail=f"Error bulk reprocessing files: {str(e)}")
@router.post("/files/{file_id}/reprocess")
@require_login
def reprocess_single_file(request: Request, file_id: int, db: Session = Depends(get_db)):
"""
Reprocess a single file by queuing it for processing again.
Args:
file_id: ID of the file to reprocess
Returns:
Task ID and status information
"""
try:
# Find the file record
file_record = db.query(FileRecord).filter(FileRecord.id == file_id).first()
if not file_record:
raise HTTPException(status_code=404, detail=f"File with ID {file_id} not found")
# Check if local file exists
if not file_record.local_filename or not os.path.exists(file_record.local_filename):
raise HTTPException(
status_code=400,
detail="Local file not found on disk. Cannot reprocess."
)
# Queue the file for processing
task = process_document.delay(file_record.local_filename, original_filename=file_record.original_filename)
logger.info(
f"Reprocessing file: ID={file_record.id}, "
f"Filename={file_record.original_filename}, TaskID={task.id}"
)
return {
"status": "success",
"message": "File queued for reprocessing",
"file_id": file_record.id,
"filename": file_record.original_filename,
"task_id": task.id
}
except HTTPException:
raise
except Exception as e:
logger.exception(f"Error reprocessing file {file_id}: {str(e)}")
raise HTTPException(status_code=500, detail=f"Error reprocessing file: {str(e)}")
@router.get("/files/{file_id}/preview")
@require_login
def get_file_preview(request: Request, file_id: int, version: str = Query("original", description="original or processed"), db: Session = Depends(get_db)):
"""
Get file content for preview (original or processed version).
Args:
file_id: ID of the file
version: "original" for tmp file, "processed" for processed file
Returns:
File content for preview
"""
from fastapi.responses import FileResponse
try:
# Find the file record
file_record = db.query(FileRecord).filter(FileRecord.id == file_id).first()
if not file_record:
raise HTTPException(status_code=404, detail=f"File with ID {file_id} not found")
if version == "original":
# Return the original file from tmp
if not file_record.local_filename or not os.path.exists(file_record.local_filename):
raise HTTPException(status_code=404, detail="Original file not found on disk")
file_path = file_record.local_filename
elif version == "processed":
# Look for processed file in /workdir/processed/
workdir = settings.workdir
processed_dir = os.path.join(workdir, "processed")
# Try to find the processed file (same hash or UUID-based naming)
base_filename = os.path.splitext(file_record.original_filename)[0]
potential_paths = [
os.path.join(processed_dir, f"{file_record.filehash}.pdf"),
os.path.join(processed_dir, f"{base_filename}_processed.pdf"),
os.path.join(processed_dir, file_record.original_filename),
]
file_path = None
for path in potential_paths:
if os.path.exists(path):
file_path = path
break
if not file_path:
raise HTTPException(status_code=404, detail="Processed file not found")
else:
raise HTTPException(status_code=400, detail="Invalid version parameter. Use 'original' or 'processed'")
# Return the file
return FileResponse(
path=file_path,
media_type=file_record.mime_type or "application/pdf",
filename=file_record.original_filename
)
except HTTPException:
raise
except Exception as e:
logger.exception(f"Error retrieving file preview: {str(e)}")
raise HTTPException(status_code=500, detail=f"Error retrieving file preview: {str(e)}")
@router.post("/ui-upload")
@require_login
async def ui_upload(request: Request, file: UploadFile = File(...)):
+86 -1
View File
@@ -7,6 +7,7 @@ from typing import Optional
from app.views.base import APIRouter, templates, require_login, get_db, logger
from app.utils.file_status import get_files_processing_status
from app.config import settings
router = APIRouter()
@@ -180,11 +181,32 @@ def file_detail_page(request: Request, file_id: int, db: Session = Depends(get_d
# Check if file exists on disk
file_exists = os.path.exists(file_record.local_filename) if file_record.local_filename else False
# Check if processed file exists
processed_exists = False
workdir = settings.workdir
processed_dir = os.path.join(workdir, "processed")
if os.path.exists(processed_dir):
base_filename = os.path.splitext(file_record.original_filename)[0]
potential_paths = [
os.path.join(processed_dir, f"{file_record.filehash}.pdf"),
os.path.join(processed_dir, f"{base_filename}_processed.pdf"),
os.path.join(processed_dir, file_record.original_filename),
]
for path in potential_paths:
if os.path.exists(path):
processed_exists = True
break
# Compute processing flow for visualization
flow_data = _compute_processing_flow(logs)
return templates.TemplateResponse("file_detail.html", {
"request": request,
"file": file_record,
"logs": logs,
"file_exists": file_exists
"file_exists": file_exists,
"processed_exists": processed_exists,
"flow_data": flow_data
})
except Exception as e:
logger.error(f"Error retrieving file details: {str(e)}")
@@ -192,3 +214,66 @@ def file_detail_page(request: Request, file_id: int, db: Session = Depends(get_d
"request": request,
"error": str(e)
})
def _compute_processing_flow(logs):
"""
Compute the processing flow structure from logs for visualization.
Returns a structured representation of the processing pipeline with branches.
"""
# Define the processing stages and their relationships
stages = {
"hash_file": {"label": "File Upload & Hash", "next": ["create_file_record"]},
"create_file_record": {"label": "Create File Record", "next": ["check_text"]},
"check_text": {"label": "Check Embedded Text", "next": ["extract_text", "process_with_azure_document_intelligence"]},
"extract_text": {"label": "Extract Text (Local)", "next": ["extract_metadata_with_gpt"]},
"process_with_azure_document_intelligence": {"label": "OCR Processing (Azure)", "next": ["extract_metadata_with_gpt"]},
"extract_metadata_with_gpt": {"label": "Extract Metadata (GPT)", "next": ["embed_metadata_into_pdf"]},
"embed_metadata_into_pdf": {"label": "Embed Metadata into PDF", "next": ["finalize_document_storage"]},
"finalize_document_storage": {"label": "Finalize & Queue Distribution", "next": ["upload_destinations"]},
"upload_destinations": {"label": "Upload to Destinations", "next": []}
}
# Create a map of step names to their log entries
step_map = {}
for log in logs:
step_name = log.step_name
if step_name not in step_map:
step_map[step_name] = []
step_map[step_name].append({
"status": log.status,
"message": log.message,
"timestamp": log.timestamp,
"task_id": log.task_id
})
# Build the flow structure
flow = []
for stage_key, stage_info in stages.items():
stage_logs = step_map.get(stage_key, [])
# Determine overall status for this stage
if stage_logs:
latest_log = stage_logs[-1]
status = latest_log["status"]
message = latest_log["message"]
timestamp = latest_log["timestamp"]
task_id = latest_log["task_id"]
else:
status = "not_run"
message = None
timestamp = None
task_id = None
flow.append({
"key": stage_key,
"label": stage_info["label"],
"status": status,
"message": message,
"timestamp": timestamp,
"task_id": task_id,
"can_retry": status == "failure"
})
return flow
+36 -3
View File
@@ -91,16 +91,49 @@ Retrieve metadata for a specific file.
**POST** `/api/files/{file_id}/reprocess`
Reprocess a specific file.
Reprocess a specific file. This queues the file for complete reprocessing through the entire pipeline.
**Response**:
```json
{
"success": true,
"message": "File queued for reprocessing"
"status": "success",
"message": "File queued for reprocessing",
"file_id": 123,
"filename": "invoice.pdf",
"task_id": "a1b2c3d4-e5f6-7g8h-9i0j-k1l2m3n4o5p6"
}
```
**Error Responses**:
- `404`: File not found
- `400`: Local file not found on disk (cannot reprocess)
### File Preview
**GET** `/api/files/{file_id}/preview`
Retrieve the file content for preview purposes.
**Parameters**:
- `version` (required): Either `original` or `processed`
- `original`: Returns the file as it was uploaded (from tmp directory)
- `processed`: Returns the file after metadata embedding (from processed directory)
**Response**: Returns the file content with appropriate MIME type for browser display.
**Example**:
```bash
# Preview original file
curl "http://<your-instance>/api/files/123/preview?version=original"
# Preview processed file
curl "http://<your-instance>/api/files/123/preview?version=processed"
```
**Error Responses**:
- `404`: File not found in database or on disk
- `400`: Invalid version parameter
### Batch Processing
**POST** `/api/processall`
+45
View File
@@ -69,6 +69,51 @@ The **Files** page provides access to all processed documents:
3. Click on any file to view its details
4. Sort the list by any column by clicking on the column header
### File Detail View
When you click on a file, you'll see a comprehensive detail view with the following sections:
#### File Information
This section displays metadata about the file:
- File ID and original filename
- File hash (SHA-256)
- File size and MIME type
- Creation timestamp
- Local path and disk status
#### Processing History
View the complete processing history with a timeline showing:
- Each processing step that was executed
- Status of each step (success, failure, in progress, pending)
- Error messages for failed steps
- Timestamps for each operation
**Retry Processing**: If a file's processing has failed, you can use the "Retry Processing" button to reprocess the entire file. This is useful when:
- External API services (like OpenAI) had temporary issues
- Network connectivity was lost during processing
- Configuration has been updated and you want to reprocess with new settings
#### Process Flow Visualization
The process flow visualization shows a visual representation of the document processing pipeline:
- **Green indicators**: Successful stages
- **Red indicators**: Failed stages with error messages
- **Gray indicators**: Stages that were not executed (e.g., because a previous stage failed)
This helps you understand:
- Which processing path your document took (local text extraction vs. OCR)
- Where exactly the processing failed
- Which stages were skipped and why
#### File Previews
If your file is still available on disk, you can preview it directly in the browser:
- **Original File**: The file as it was uploaded
- **Processed File**: The file after metadata has been embedded (if processing completed)
Both previews support:
- In-browser PDF viewing
- Opening in a new tab for full-screen viewing
- Side-by-side comparison of original and processed versions
## Document Processing Features
Depending on the system configuration, DocuElevate can perform:
+151 -1
View File
@@ -207,6 +207,57 @@
color: #991B1B;
}
</style>
<script>
// JavaScript for handling retry functionality
async function reprocessFile() {
const fileId = {{ file.id | tojson }};
const button = document.getElementById('reprocess-btn');
const statusDiv = document.getElementById('reprocess-status');
// Disable button and show loading
button.disabled = true;
button.innerHTML = '<i class="fas fa-spinner fa-spin"></i> Processing...';
statusDiv.innerHTML = '';
try {
const response = await fetch(`/api/files/${fileId}/reprocess`, {
method: 'POST',
headers: {
'Content-Type': 'application/json'
}
});
const data = await response.json();
if (response.ok) {
statusDiv.innerHTML = `
<div style="background-color: #D1FAE5; color: #065F46; padding: 1rem; border-radius: 0.25rem; margin-top: 1rem;">
<i class="fas fa-check-circle"></i> File queued for reprocessing. Task ID: ${data.task_id}
<br><small>Refresh the page in a few moments to see updated status.</small>
</div>
`;
// Reload page after 3 seconds
setTimeout(() => window.location.reload(), 3000);
} else {
statusDiv.innerHTML = `
<div class="error-message" style="margin-top: 1rem;">
<i class="fas fa-exclamation-triangle"></i> Error: ${data.detail || 'Failed to reprocess file'}
</div>
`;
button.disabled = false;
button.innerHTML = '<i class="fas fa-redo"></i> Retry Processing';
}
} catch (error) {
statusDiv.innerHTML = `
<div class="error-message" style="margin-top: 1rem;">
<i class="fas fa-exclamation-triangle"></i> Network error: ${error.message}
</div>
`;
button.disabled = false;
button.innerHTML = '<i class="fas fa-redo"></i> Retry Processing';
}
}
</script>
{% endblock %}
{% block content %}
@@ -273,7 +324,15 @@
<!-- Processing History Card -->
<div class="detail-card">
<h3>Processing History</h3>
<div style="display: flex; justify-content: space-between; align-items: center; margin-bottom: 1rem;">
<h3 style="margin: 0;">Processing History</h3>
{% if file_exists %}
<button id="reprocess-btn" onclick="reprocessFile()" style="background-color: #3182ce; color: white; padding: 0.5rem 1rem; border: none; border-radius: 0.25rem; cursor: pointer; font-weight: 600;">
<i class="fas fa-redo"></i> Retry Processing
</button>
{% endif %}
</div>
<div id="reprocess-status"></div>
{% if logs %}
<div class="timeline">
{% for log in logs %}
@@ -305,6 +364,97 @@
{% endif %}
</div>
<!-- Process Flow Visualization Card -->
{% if flow_data %}
<div class="detail-card">
<h3>Process Flow Visualization</h3>
<div style="padding: 1rem 0;">
<div style="position: relative;">
{% for stage in flow_data %}
<div style="display: flex; align-items: center; margin-bottom: 1.5rem;">
<!-- Status indicator -->
<div style="width: 40px; height: 40px; border-radius: 50%; display: flex; align-items: center; justify-content: center; margin-right: 1rem;
{% if stage.status == 'success' %}background-color: #48bb78; color: white;
{% elif stage.status == 'failure' %}background-color: #f56565; color: white;
{% elif stage.status == 'in_progress' %}background-color: #4299e1; color: white;
{% elif stage.status == 'pending' %}background-color: #ecc94b; color: white;
{% else %}background-color: #e2e8f0; color: #718096;{% endif %}
">
{% if stage.status == 'success' %}<i class="fas fa-check"></i>
{% elif stage.status == 'failure' %}<i class="fas fa-times"></i>
{% elif stage.status == 'in_progress' %}<i class="fas fa-spinner fa-spin"></i>
{% elif stage.status == 'pending' %}<i class="fas fa-clock"></i>
{% else %}<i class="fas fa-circle"></i>{% endif %}
</div>
<!-- Stage info -->
<div style="flex: 1; background-color: #f7fafc; padding: 1rem; border-radius: 0.5rem; border-left: 3px solid
{% if stage.status == 'success' %}#48bb78
{% elif stage.status == 'failure' %}#f56565
{% elif stage.status == 'in_progress' %}#4299e1
{% elif stage.status == 'pending' %}#ecc94b
{% else %}#e2e8f0{% endif %};">
<div style="font-weight: 600; color: #2d3748; margin-bottom: 0.25rem;">{{ stage.label }}</div>
{% if stage.message %}
<div style="color: #4a5568; font-size: 0.875rem;">{{ stage.message }}</div>
{% endif %}
{% if stage.status == 'not_run' %}
<div style="color: #718096; font-size: 0.875rem; font-style: italic;">Not executed</div>
{% endif %}
{% if stage.timestamp %}
<div style="color: #718096; font-size: 0.75rem; margin-top: 0.25rem;">
{{ stage.timestamp.strftime('%Y-%m-%d %H:%M:%S') }}
</div>
{% endif %}
</div>
</div>
<!-- Connector line (except for last item) -->
{% if not loop.last %}
<div style="width: 2px; height: 20px; background-color: #e2e8f0; margin-left: 19px; margin-bottom: 0.5rem;"></div>
{% endif %}
{% endfor %}
</div>
</div>
</div>
{% endif %}
<!-- File Preview Card -->
{% if file_exists or processed_exists %}
<div class="detail-card">
<h3>File Previews</h3>
<div style="display: grid; grid-template-columns: repeat(auto-fit, minmax(400px, 1fr)); gap: 2rem;">
{% if file_exists %}
<div>
<h4 style="font-weight: 600; color: #2d3748; margin-bottom: 1rem;">Original File</h4>
<div style="border: 2px solid #e2e8f0; border-radius: 0.5rem; overflow: hidden; background-color: #f7fafc;">
<embed src="/api/files/{{ file.id }}/preview?version=original" type="application/pdf" width="100%" height="600px" style="border: none;">
</div>
<div style="margin-top: 0.5rem; text-align: center;">
<a href="/api/files/{{ file.id }}/preview?version=original" target="_blank" style="color: #3182ce; text-decoration: none; font-size: 0.875rem;">
<i class="fas fa-external-link-alt"></i> Open in new tab
</a>
</div>
</div>
{% endif %}
{% if processed_exists %}
<div>
<h4 style="font-weight: 600; color: #2d3748; margin-bottom: 1rem;">Processed File</h4>
<div style="border: 2px solid #e2e8f0; border-radius: 0.5rem; overflow: hidden; background-color: #f7fafc;">
<embed src="/api/files/{{ file.id }}/preview?version=processed" type="application/pdf" width="100%" height="600px" style="border: none;">
</div>
<div style="margin-top: 0.5rem; text-align: center;">
<a href="/api/files/{{ file.id }}/preview?version=processed" target="_blank" style="color: #3182ce; text-decoration: none; font-size: 0.875rem;">
<i class="fas fa-external-link-alt"></i> Open in new tab
</a>
</div>
</div>
{% endif %}
</div>
</div>
{% endif %}
{% endif %}
</div>
{% endblock %}
+200
View File
@@ -0,0 +1,200 @@
"""
Tests for file detail view improvements including reprocessing and preview endpoints.
"""
import os
import pytest
from fastapi.testclient import TestClient
from app.models import FileRecord, ProcessingLog
@pytest.mark.integration
class TestFileReprocessing:
"""Tests for single file reprocessing endpoint."""
def test_reprocess_existing_file(self, client: TestClient, db_session, sample_pdf_path):
"""Test reprocessing an existing file."""
# Create a file record
file_record = FileRecord(
filehash="abc123",
original_filename="test.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf"
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Add a failed processing log
log = ProcessingLog(
file_id=file_record.id,
task_id="test-task-123",
step_name="extract_metadata_with_gpt",
status="failure",
message="API error"
)
db_session.add(log)
db_session.commit()
# Test reprocessing
response = client.post(f"/api/files/{file_record.id}/reprocess")
assert response.status_code == 200
data = response.json()
assert data["status"] == "success"
assert "task_id" in data
assert data["file_id"] == file_record.id
assert data["filename"] == "test.pdf"
def test_reprocess_nonexistent_file(self, client: TestClient):
"""Test reprocessing a file that doesn't exist."""
response = client.post("/api/files/99999/reprocess")
assert response.status_code == 404
assert "not found" in response.json()["detail"].lower()
def test_reprocess_file_missing_on_disk(self, client: TestClient, db_session):
"""Test reprocessing when local file is missing."""
# Create a file record with non-existent local path
file_record = FileRecord(
filehash="xyz789",
original_filename="missing.pdf",
local_filename="/nonexistent/path/missing.pdf",
file_size=1024,
mime_type="application/pdf"
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Test reprocessing
response = client.post(f"/api/files/{file_record.id}/reprocess")
assert response.status_code == 400
assert "not found on disk" in response.json()["detail"].lower()
@pytest.mark.integration
class TestFilePreview:
"""Tests for file preview endpoint."""
def test_preview_original_file(self, client: TestClient, db_session, sample_pdf_path):
"""Test getting original file preview."""
# Create a file record
file_record = FileRecord(
filehash="def456",
original_filename="preview.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf"
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Test preview
response = client.get(f"/api/files/{file_record.id}/preview?version=original")
assert response.status_code == 200
assert response.headers["content-type"].startswith("application/pdf")
def test_preview_processed_file_not_found(self, client: TestClient, db_session, sample_pdf_path):
"""Test getting processed file preview when it doesn't exist."""
# Create a file record
file_record = FileRecord(
filehash="ghi789",
original_filename="processed.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf"
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Test preview (processed version should not exist)
response = client.get(f"/api/files/{file_record.id}/preview?version=processed")
assert response.status_code == 404
assert "not found" in response.json()["detail"].lower()
def test_preview_nonexistent_file(self, client: TestClient):
"""Test preview for a file that doesn't exist."""
response = client.get("/api/files/99999/preview?version=original")
assert response.status_code == 404
assert "not found" in response.json()["detail"].lower()
def test_preview_invalid_version(self, client: TestClient, db_session, sample_pdf_path):
"""Test preview with invalid version parameter."""
# Create a file record
file_record = FileRecord(
filehash="jkl012",
original_filename="test.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf"
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Test with invalid version
response = client.get(f"/api/files/{file_record.id}/preview?version=invalid")
assert response.status_code == 400
assert "invalid version" in response.json()["detail"].lower()
@pytest.mark.integration
class TestFileDetailView:
"""Tests for enhanced file detail view."""
def test_file_detail_view_with_logs(self, client: TestClient, db_session, sample_pdf_path):
"""Test file detail view returns enhanced data."""
# Create a file record
file_record = FileRecord(
filehash="mno345",
original_filename="detail.pdf",
local_filename=sample_pdf_path,
file_size=1024,
mime_type="application/pdf"
)
db_session.add(file_record)
db_session.commit()
db_session.refresh(file_record)
# Add processing logs
logs = [
ProcessingLog(
file_id=file_record.id,
task_id="task-1",
step_name="hash_file",
status="success",
message="File hashed successfully"
),
ProcessingLog(
file_id=file_record.id,
task_id="task-1",
step_name="create_file_record",
status="success",
message="File record created"
),
ProcessingLog(
file_id=file_record.id,
task_id="task-1",
step_name="extract_metadata_with_gpt",
status="failure",
message="API rate limit exceeded"
)
]
for log in logs:
db_session.add(log)
db_session.commit()
# Test detail view
response = client.get(f"/files/{file_record.id}/detail")
assert response.status_code == 200
# Check that response contains HTML with file information
assert b"File Information" in response.content
assert b"detail.pdf" in response.content
assert b"Processing History" in response.content
def test_file_detail_view_nonexistent(self, client: TestClient):
"""Test file detail view for nonexistent file."""
response = client.get("/files/99999/detail")
assert response.status_code == 200 # Returns page with error message
assert b"not found" in response.content.lower()