From 76ab69810511aa1f2d5d76995ab55194e2031905 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Sat, 7 Feb 2026 20:52:08 +0000 Subject: [PATCH] Add backend endpoints and enhanced file detail view - Added /api/files/{file_id}/reprocess endpoint for single file reprocessing - Added /api/files/{file_id}/preview endpoint for viewing original/processed files - Enhanced file detail view with process flow computation - Updated frontend template with retry button, process flow visualization, and PDF previews - Added JavaScript for async retry functionality Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com> --- app/api/files.py | 116 +++++++++++++++++++++ app/views/files.py | 87 +++++++++++++++- frontend/templates/file_detail.html | 152 +++++++++++++++++++++++++++- 3 files changed, 353 insertions(+), 2 deletions(-) diff --git a/app/api/files.py b/app/api/files.py index 4f28fa6a..94106db5 100644 --- a/app/api/files.py +++ b/app/api/files.py @@ -363,6 +363,122 @@ def bulk_reprocess_files(request: Request, file_ids: List[int], db: Session = De raise HTTPException(status_code=500, detail=f"Error bulk reprocessing files: {str(e)}") +@router.post("/files/{file_id}/reprocess") +@require_login +def reprocess_single_file(request: Request, file_id: int, db: Session = Depends(get_db)): + """ + Reprocess a single file by queuing it for processing again. + + Args: + file_id: ID of the file to reprocess + + Returns: + Task ID and status information + """ + try: + # Find the file record + file_record = db.query(FileRecord).filter(FileRecord.id == file_id).first() + + if not file_record: + raise HTTPException(status_code=404, detail=f"File with ID {file_id} not found") + + # Check if local file exists + if not file_record.local_filename or not os.path.exists(file_record.local_filename): + raise HTTPException( + status_code=400, + detail="Local file not found on disk. Cannot reprocess." + ) + + # Queue the file for processing + task = process_document.delay(file_record.local_filename, original_filename=file_record.original_filename) + + logger.info( + f"Reprocessing file: ID={file_record.id}, " + f"Filename={file_record.original_filename}, TaskID={task.id}" + ) + + return { + "status": "success", + "message": "File queued for reprocessing", + "file_id": file_record.id, + "filename": file_record.original_filename, + "task_id": task.id + } + + except HTTPException: + raise + except Exception as e: + logger.exception(f"Error reprocessing file {file_id}: {str(e)}") + raise HTTPException(status_code=500, detail=f"Error reprocessing file: {str(e)}") + + +@router.get("/files/{file_id}/preview") +@require_login +def get_file_preview(request: Request, file_id: int, version: str = Query("original", description="original or processed"), db: Session = Depends(get_db)): + """ + Get file content for preview (original or processed version). + + Args: + file_id: ID of the file + version: "original" for tmp file, "processed" for processed file + + Returns: + File content for preview + """ + from fastapi.responses import FileResponse + + try: + # Find the file record + file_record = db.query(FileRecord).filter(FileRecord.id == file_id).first() + + if not file_record: + raise HTTPException(status_code=404, detail=f"File with ID {file_id} not found") + + if version == "original": + # Return the original file from tmp + if not file_record.local_filename or not os.path.exists(file_record.local_filename): + raise HTTPException(status_code=404, detail="Original file not found on disk") + + file_path = file_record.local_filename + + elif version == "processed": + # Look for processed file in /workdir/processed/ + workdir = settings.workdir + processed_dir = os.path.join(workdir, "processed") + + # Try to find the processed file (same hash or UUID-based naming) + base_filename = os.path.splitext(file_record.original_filename)[0] + potential_paths = [ + os.path.join(processed_dir, f"{file_record.filehash}.pdf"), + os.path.join(processed_dir, f"{base_filename}_processed.pdf"), + os.path.join(processed_dir, file_record.original_filename), + ] + + file_path = None + for path in potential_paths: + if os.path.exists(path): + file_path = path + break + + if not file_path: + raise HTTPException(status_code=404, detail="Processed file not found") + else: + raise HTTPException(status_code=400, detail="Invalid version parameter. Use 'original' or 'processed'") + + # Return the file + return FileResponse( + path=file_path, + media_type=file_record.mime_type or "application/pdf", + filename=file_record.original_filename + ) + + except HTTPException: + raise + except Exception as e: + logger.exception(f"Error retrieving file preview: {str(e)}") + raise HTTPException(status_code=500, detail=f"Error retrieving file preview: {str(e)}") + + @router.post("/ui-upload") @require_login async def ui_upload(request: Request, file: UploadFile = File(...)): diff --git a/app/views/files.py b/app/views/files.py index ea793f33..10298c81 100644 --- a/app/views/files.py +++ b/app/views/files.py @@ -7,6 +7,7 @@ from typing import Optional from app.views.base import APIRouter, templates, require_login, get_db, logger from app.utils.file_status import get_files_processing_status +from app.config import settings router = APIRouter() @@ -180,11 +181,32 @@ def file_detail_page(request: Request, file_id: int, db: Session = Depends(get_d # Check if file exists on disk file_exists = os.path.exists(file_record.local_filename) if file_record.local_filename else False + # Check if processed file exists + processed_exists = False + workdir = settings.workdir + processed_dir = os.path.join(workdir, "processed") + if os.path.exists(processed_dir): + base_filename = os.path.splitext(file_record.original_filename)[0] + potential_paths = [ + os.path.join(processed_dir, f"{file_record.filehash}.pdf"), + os.path.join(processed_dir, f"{base_filename}_processed.pdf"), + os.path.join(processed_dir, file_record.original_filename), + ] + for path in potential_paths: + if os.path.exists(path): + processed_exists = True + break + + # Compute processing flow for visualization + flow_data = _compute_processing_flow(logs) + return templates.TemplateResponse("file_detail.html", { "request": request, "file": file_record, "logs": logs, - "file_exists": file_exists + "file_exists": file_exists, + "processed_exists": processed_exists, + "flow_data": flow_data }) except Exception as e: logger.error(f"Error retrieving file details: {str(e)}") @@ -192,3 +214,66 @@ def file_detail_page(request: Request, file_id: int, db: Session = Depends(get_d "request": request, "error": str(e) }) + + +def _compute_processing_flow(logs): + """ + Compute the processing flow structure from logs for visualization. + + Returns a structured representation of the processing pipeline with branches. + """ + # Define the processing stages and their relationships + stages = { + "hash_file": {"label": "File Upload & Hash", "next": ["create_file_record"]}, + "create_file_record": {"label": "Create File Record", "next": ["check_text"]}, + "check_text": {"label": "Check Embedded Text", "next": ["extract_text", "process_with_azure_document_intelligence"]}, + "extract_text": {"label": "Extract Text (Local)", "next": ["extract_metadata_with_gpt"]}, + "process_with_azure_document_intelligence": {"label": "OCR Processing (Azure)", "next": ["extract_metadata_with_gpt"]}, + "extract_metadata_with_gpt": {"label": "Extract Metadata (GPT)", "next": ["embed_metadata_into_pdf"]}, + "embed_metadata_into_pdf": {"label": "Embed Metadata into PDF", "next": ["finalize_document_storage"]}, + "finalize_document_storage": {"label": "Finalize & Queue Distribution", "next": ["upload_destinations"]}, + "upload_destinations": {"label": "Upload to Destinations", "next": []} + } + + # Create a map of step names to their log entries + step_map = {} + for log in logs: + step_name = log.step_name + if step_name not in step_map: + step_map[step_name] = [] + step_map[step_name].append({ + "status": log.status, + "message": log.message, + "timestamp": log.timestamp, + "task_id": log.task_id + }) + + # Build the flow structure + flow = [] + for stage_key, stage_info in stages.items(): + stage_logs = step_map.get(stage_key, []) + + # Determine overall status for this stage + if stage_logs: + latest_log = stage_logs[-1] + status = latest_log["status"] + message = latest_log["message"] + timestamp = latest_log["timestamp"] + task_id = latest_log["task_id"] + else: + status = "not_run" + message = None + timestamp = None + task_id = None + + flow.append({ + "key": stage_key, + "label": stage_info["label"], + "status": status, + "message": message, + "timestamp": timestamp, + "task_id": task_id, + "can_retry": status == "failure" + }) + + return flow diff --git a/frontend/templates/file_detail.html b/frontend/templates/file_detail.html index 96153a5f..dcae4510 100644 --- a/frontend/templates/file_detail.html +++ b/frontend/templates/file_detail.html @@ -207,6 +207,57 @@ color: #991B1B; } + {% endblock %} {% block content %} @@ -273,7 +324,15 @@