feat(ui): add verbose worker log detail to processing history on file detail page
Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
@@ -48,11 +48,33 @@ def init_db():
|
||||
try:
|
||||
Base.metadata.create_all(bind=engine)
|
||||
logger.info("Database initialization complete (tables created if not exist).")
|
||||
|
||||
# 6. Run lightweight schema migrations for existing databases
|
||||
_run_schema_migrations(engine)
|
||||
except exc.SQLAlchemyError as e:
|
||||
logger.error(f"Error initializing database: {e}")
|
||||
raise
|
||||
|
||||
|
||||
def _run_schema_migrations(engine):
|
||||
"""
|
||||
Apply lightweight schema migrations for columns added after the initial release.
|
||||
Each migration is idempotent and safe to run multiple times.
|
||||
"""
|
||||
from sqlalchemy import inspect, text
|
||||
|
||||
inspector = inspect(engine)
|
||||
|
||||
# Migration: Add 'detail' column to processing_logs (added for verbose worker log output)
|
||||
if "processing_logs" in inspector.get_table_names():
|
||||
columns = [col["name"] for col in inspector.get_columns("processing_logs")]
|
||||
if "detail" not in columns:
|
||||
logger.info("Migrating processing_logs: adding 'detail' column")
|
||||
with engine.begin() as conn:
|
||||
conn.execute(text("ALTER TABLE processing_logs ADD COLUMN detail TEXT"))
|
||||
logger.info("Migration complete: 'detail' column added to processing_logs")
|
||||
|
||||
|
||||
def get_db():
|
||||
"""
|
||||
Dependency for FastAPI routes or general DB usage.
|
||||
|
||||
+2
-1
@@ -1,6 +1,6 @@
|
||||
# app/models.py
|
||||
|
||||
from sqlalchemy import Column, DateTime, ForeignKey, Integer, String, func
|
||||
from sqlalchemy import Column, DateTime, ForeignKey, Integer, String, Text, func
|
||||
|
||||
from app.database import Base
|
||||
|
||||
@@ -48,6 +48,7 @@ class ProcessingLog(Base):
|
||||
step_name = Column(String) # e.g., "OCR", "convert_to_pdf", "upload_s3"
|
||||
status = Column(String) # "pending", "in_progress", "success", "failure"
|
||||
message = Column(String, nullable=True) # Error text or success note
|
||||
detail = Column(Text, nullable=True) # Verbose worker log output for diagnostics
|
||||
timestamp = Column(DateTime(timezone=True), server_default=func.now())
|
||||
|
||||
|
||||
|
||||
@@ -98,7 +98,14 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met
|
||||
local_file_path = alt_path
|
||||
else:
|
||||
logger.error(f"[{task_id}] Local file {local_file_path} not found, cannot embed metadata.")
|
||||
log_task_progress(task_id, "embed_metadata_into_pdf", "failure", "File not found", file_id=file_id)
|
||||
log_task_progress(
|
||||
task_id, "embed_metadata_into_pdf", "failure", "File not found", file_id=file_id,
|
||||
detail=(
|
||||
f"Local file not found, cannot embed metadata.\n"
|
||||
f"Tried path: {local_file_path}\n"
|
||||
f"Also tried: {alt_path}"
|
||||
),
|
||||
)
|
||||
return {"error": "File not found"}
|
||||
|
||||
# Work on a safe copy in a secure temporary directory
|
||||
@@ -184,7 +191,14 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met
|
||||
# Trigger the next step: final storage.
|
||||
logger.info(f"[{task_id}] Queueing final storage task")
|
||||
log_task_progress(
|
||||
task_id, "embed_metadata_into_pdf", "success", "Metadata embedded, queuing finalization", file_id=file_id
|
||||
task_id, "embed_metadata_into_pdf", "success", "Metadata embedded, queuing finalization", file_id=file_id,
|
||||
detail=(
|
||||
f"Metadata embedded into PDF successfully.\n"
|
||||
f"Original file: {original_file}\n"
|
||||
f"Final file: {final_file_path}\n"
|
||||
f"Metadata JSON: {json_path}\n"
|
||||
f"Suggested filename: {suggested_filename}.pdf"
|
||||
),
|
||||
)
|
||||
finalize_document_storage.delay(original_file, final_file_path, metadata, file_id=file_id)
|
||||
|
||||
@@ -209,7 +223,10 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"[{task_id}] Failed to embed metadata into {processed_file}: {e}")
|
||||
log_task_progress(task_id, "embed_metadata_into_pdf", "failure", f"Exception: {str(e)}", file_id=file_id)
|
||||
log_task_progress(
|
||||
task_id, "embed_metadata_into_pdf", "failure", f"Exception: {str(e)}", file_id=file_id,
|
||||
detail=f"Failed to embed metadata into {processed_file}.\nOriginal file: {original_file}\nException: {str(e)}",
|
||||
)
|
||||
# Clean up temporary file in case of error
|
||||
if os.path.exists(processed_file):
|
||||
try:
|
||||
|
||||
@@ -111,13 +111,17 @@ def extract_metadata_with_gpt(self, filename: str, cleaned_text: str, file_id: i
|
||||
|
||||
content = completion.choices[0].message.content
|
||||
logger.info(f"[{task_id}] Raw classification response for {filename}: {content[:200]}...")
|
||||
log_task_progress(task_id, "call_openai", "success", "Received OpenAI response", file_id=file_id)
|
||||
log_task_progress(
|
||||
task_id, "call_openai", "success", "Received OpenAI response", file_id=file_id,
|
||||
detail=f"Raw classification response:\n{content}",
|
||||
)
|
||||
|
||||
json_text = extract_json_from_text(content)
|
||||
if not json_text:
|
||||
logger.error(f"[{task_id}] Could not find valid JSON in GPT response for {filename}.")
|
||||
log_task_progress(
|
||||
task_id, "extract_metadata_with_gpt", "failure", "Invalid JSON in response", file_id=file_id
|
||||
task_id, "extract_metadata_with_gpt", "failure", "Invalid JSON in response", file_id=file_id,
|
||||
detail=f"Could not parse valid JSON from GPT response.\nRaw response:\n{content}",
|
||||
)
|
||||
return {}
|
||||
|
||||
@@ -143,7 +147,8 @@ def extract_metadata_with_gpt(self, filename: str, cleaned_text: str, file_id: i
|
||||
|
||||
logger.info(f"[{task_id}] Extracted metadata: {metadata}")
|
||||
log_task_progress(
|
||||
task_id, "parse_metadata", "success", f"Parsed metadata: {list(metadata.keys())}", file_id=file_id
|
||||
task_id, "parse_metadata", "success", f"Parsed metadata: {list(metadata.keys())}", file_id=file_id,
|
||||
detail=f"Extracted metadata:\n{json.dumps(metadata, ensure_ascii=False, indent=2)}",
|
||||
)
|
||||
|
||||
# Trigger the next step: embedding metadata into the PDF
|
||||
@@ -158,5 +163,8 @@ def extract_metadata_with_gpt(self, filename: str, cleaned_text: str, file_id: i
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"[{task_id}] OpenAI classification failed for {filename}: {e}")
|
||||
log_task_progress(task_id, "extract_metadata_with_gpt", "failure", f"Exception: {str(e)}", file_id=file_id)
|
||||
log_task_progress(
|
||||
task_id, "extract_metadata_with_gpt", "failure", f"Exception: {str(e)}", file_id=file_id,
|
||||
detail=f"OpenAI classification failed for {filename}.\nException: {str(e)}",
|
||||
)
|
||||
return {}
|
||||
|
||||
@@ -52,7 +52,10 @@ def process_document(self, original_local_file: str, original_filename: str = No
|
||||
|
||||
if not os.path.exists(original_local_file):
|
||||
logger.error(f"[{task_id}] File {original_local_file} not found.")
|
||||
log_task_progress(task_id, "process_document", "failure", "File not found")
|
||||
log_task_progress(
|
||||
task_id, "process_document", "failure", "File not found",
|
||||
detail=f"File not found on disk: {original_local_file}",
|
||||
)
|
||||
return {"error": "File not found"}
|
||||
|
||||
# 0. Compute the file hash and check for duplicates
|
||||
@@ -104,6 +107,12 @@ def process_document(self, original_local_file: str, original_filename: str = No
|
||||
"success",
|
||||
"Duplicate file detected, skipping",
|
||||
file_id=existing.id,
|
||||
detail=(
|
||||
f"Duplicate file detected.\n"
|
||||
f"File hash: {filehash}\n"
|
||||
f"Existing file record ID: {existing.id}\n"
|
||||
f"Original filename: {original_filename}"
|
||||
),
|
||||
)
|
||||
return {
|
||||
"status": "duplicate_file",
|
||||
|
||||
@@ -245,13 +245,17 @@ def upload_to_paperless(self, file_path: str, file_id: int = None):
|
||||
resp.raise_for_status()
|
||||
except requests.exceptions.RequestException as exc:
|
||||
error_msg = f"Failed to upload to Paperless: {exc}"
|
||||
response_text = getattr(exc.response, "text", "<no response>")
|
||||
logger.error(
|
||||
f"[{task_id}] Failed to upload document '%s' to Paperless. Error: %s. Response=%s",
|
||||
file_path,
|
||||
exc,
|
||||
getattr(exc.response, "text", "<no response>"),
|
||||
response_text,
|
||||
)
|
||||
log_task_progress(
|
||||
task_id, "upload_to_paperless", "failure", error_msg, file_id=file_id,
|
||||
detail=f"Failed to upload document to Paperless.\nFile: {file_path}\nError: {exc}\nResponse: {response_text}",
|
||||
)
|
||||
log_task_progress(task_id, "upload_to_paperless", "failure", error_msg, file_id=file_id)
|
||||
raise
|
||||
|
||||
raw_task_id = resp.text.strip().strip('"').strip("'")
|
||||
|
||||
+10
-1
@@ -2,9 +2,17 @@ from app.database import SessionLocal
|
||||
from app.models import ProcessingLog
|
||||
|
||||
|
||||
def log_task_progress(task_id, step_name, status, message=None, file_id=None):
|
||||
def log_task_progress(task_id, step_name, status, message=None, file_id=None, detail=None):
|
||||
"""
|
||||
Logs the progress of a Celery task to the database.
|
||||
|
||||
Args:
|
||||
task_id: The Celery task ID
|
||||
step_name: Name of the processing step
|
||||
status: Current status (pending, in_progress, success, failure)
|
||||
message: Short summary message
|
||||
file_id: Optional associated file record ID
|
||||
detail: Optional verbose log output for diagnostics
|
||||
"""
|
||||
with SessionLocal() as db:
|
||||
log_entry = ProcessingLog(
|
||||
@@ -13,6 +21,7 @@ def log_task_progress(task_id, step_name, status, message=None, file_id=None):
|
||||
status=status,
|
||||
message=message,
|
||||
file_id=file_id,
|
||||
detail=detail,
|
||||
)
|
||||
db.add(log_entry)
|
||||
db.commit()
|
||||
|
||||
@@ -256,6 +256,38 @@
|
||||
max-height: 5000px;
|
||||
transition: max-height 0.5s ease-in;
|
||||
}
|
||||
|
||||
/* Verbose detail block */
|
||||
.timeline-detail-toggle {
|
||||
cursor: pointer;
|
||||
color: #3182ce;
|
||||
font-size: 0.75rem;
|
||||
font-weight: 600;
|
||||
margin-top: 0.25rem;
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 0.25rem;
|
||||
}
|
||||
.timeline-detail-toggle:hover {
|
||||
color: #2c5aa0;
|
||||
}
|
||||
.timeline-detail {
|
||||
display: none;
|
||||
margin-top: 0.5rem;
|
||||
background-color: #1a202c;
|
||||
color: #e2e8f0;
|
||||
padding: 0.75rem;
|
||||
border-radius: 0.375rem;
|
||||
font-family: monospace;
|
||||
font-size: 0.75rem;
|
||||
white-space: pre-wrap;
|
||||
word-break: break-word;
|
||||
max-height: 400px;
|
||||
overflow-y: auto;
|
||||
}
|
||||
.timeline-detail.visible {
|
||||
display: block;
|
||||
}
|
||||
|
||||
.no-logs {
|
||||
text-align: center;
|
||||
@@ -590,6 +622,21 @@
|
||||
toggleIcon.classList.add('fa-chevron-up');
|
||||
}
|
||||
}
|
||||
|
||||
// Toggle verbose detail for a log entry
|
||||
function toggleDetail(logId) {
|
||||
const detail = document.getElementById('detail-' + logId);
|
||||
const icon = document.getElementById('detail-icon-' + logId);
|
||||
if (detail.classList.contains('visible')) {
|
||||
detail.classList.remove('visible');
|
||||
icon.classList.remove('fa-chevron-up');
|
||||
icon.classList.add('fa-chevron-right');
|
||||
} else {
|
||||
detail.classList.add('visible');
|
||||
icon.classList.remove('fa-chevron-right');
|
||||
icon.classList.add('fa-chevron-up');
|
||||
}
|
||||
}
|
||||
</script>
|
||||
{% endif %}
|
||||
{% endblock %}
|
||||
@@ -760,6 +807,13 @@
|
||||
<span class="timeline-task-id">Task: {{ log.task_id[:16] }}...</span>
|
||||
{% endif %}
|
||||
</div>
|
||||
{% if log.detail %}
|
||||
<div class="timeline-detail-toggle" onclick="toggleDetail({{ loop.index }})">
|
||||
<i id="detail-icon-{{ loop.index }}" class="fas fa-chevron-right"></i>
|
||||
<span>Show worker log detail</span>
|
||||
</div>
|
||||
<div id="detail-{{ loop.index }}" class="timeline-detail">{{ log.detail }}</div>
|
||||
{% endif %}
|
||||
</div>
|
||||
</div>
|
||||
{% endfor %}
|
||||
|
||||
@@ -70,3 +70,77 @@ class TestGetDb:
|
||||
session = next(gen)
|
||||
# Force the generator to close
|
||||
gen.close()
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestSchemaMigrations:
|
||||
"""Tests for schema migration logic."""
|
||||
|
||||
def test_processing_log_detail_column_exists(self, db_session):
|
||||
"""Test that ProcessingLog has the detail column."""
|
||||
from app.models import ProcessingLog
|
||||
|
||||
log = ProcessingLog(
|
||||
task_id="test-task",
|
||||
step_name="test_step",
|
||||
status="success",
|
||||
message="Short message",
|
||||
detail="Verbose worker log output\nWith multiple lines",
|
||||
)
|
||||
db_session.add(log)
|
||||
db_session.commit()
|
||||
db_session.refresh(log)
|
||||
|
||||
assert log.detail == "Verbose worker log output\nWith multiple lines"
|
||||
|
||||
def test_processing_log_detail_nullable(self, db_session):
|
||||
"""Test that detail column is nullable (backward compatible)."""
|
||||
from app.models import ProcessingLog
|
||||
|
||||
log = ProcessingLog(
|
||||
task_id="test-task-2",
|
||||
step_name="test_step",
|
||||
status="success",
|
||||
message="Short message",
|
||||
)
|
||||
db_session.add(log)
|
||||
db_session.commit()
|
||||
db_session.refresh(log)
|
||||
|
||||
assert log.detail is None
|
||||
|
||||
def test_migration_adds_detail_column(self, tmp_path):
|
||||
"""Test that _run_schema_migrations adds detail column to existing tables."""
|
||||
from sqlalchemy import Column, Integer, String, create_engine, text
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
|
||||
from app.database import _run_schema_migrations
|
||||
|
||||
# Create a database with the old schema (no detail column)
|
||||
db_path = str(tmp_path / "migration_test.db")
|
||||
engine = create_engine(f"sqlite:///{db_path}")
|
||||
with engine.begin() as conn:
|
||||
conn.execute(
|
||||
text(
|
||||
"CREATE TABLE processing_logs ("
|
||||
"id INTEGER PRIMARY KEY, "
|
||||
"file_id INTEGER, "
|
||||
"task_id VARCHAR, "
|
||||
"step_name VARCHAR, "
|
||||
"status VARCHAR, "
|
||||
"message VARCHAR, "
|
||||
"timestamp DATETIME)"
|
||||
)
|
||||
)
|
||||
|
||||
# Run migrations
|
||||
_run_schema_migrations(engine)
|
||||
|
||||
# Verify detail column was added
|
||||
from sqlalchemy import inspect
|
||||
|
||||
inspector = inspect(engine)
|
||||
columns = [col["name"] for col in inspector.get_columns("processing_logs")]
|
||||
assert "detail" in columns
|
||||
|
||||
engine.dispose()
|
||||
|
||||
@@ -42,6 +42,7 @@ class TestTaskLogging:
|
||||
status="started",
|
||||
message="Processing document",
|
||||
file_id=456,
|
||||
detail=None,
|
||||
)
|
||||
|
||||
# Verify database operations
|
||||
@@ -67,7 +68,7 @@ class TestTaskLogging:
|
||||
|
||||
# Verify called with None for optional parameters
|
||||
mock_processing_log.assert_called_once_with(
|
||||
task_id="task-456", step_name="upload", status="completed", message=None, file_id=None
|
||||
task_id="task-456", step_name="upload", status="completed", message=None, file_id=None, detail=None
|
||||
)
|
||||
|
||||
mock_db.add.assert_called_once()
|
||||
@@ -127,6 +128,7 @@ class TestTaskLogging:
|
||||
status="success",
|
||||
message="Document processed successfully",
|
||||
file_id=999,
|
||||
detail=None,
|
||||
)
|
||||
|
||||
mock_db.add.assert_called_once()
|
||||
|
||||
Reference in New Issue
Block a user