style: fix linting issues in retry and metadata extraction code
Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
+11
-12
@@ -525,7 +525,7 @@ def _retry_pipeline_step(file_record: FileRecord, step_name: str, db: Session) -
|
|||||||
# Retrying embed requires re-running metadata extraction first, because
|
# Retrying embed requires re-running metadata extraction first, because
|
||||||
# embed_metadata_into_pdf needs the actual metadata dict (not empty).
|
# embed_metadata_into_pdf needs the actual metadata dict (not empty).
|
||||||
# Re-trigger extract_metadata_with_gpt which will chain into embed_metadata_into_pdf.
|
# Re-trigger extract_metadata_with_gpt which will chain into embed_metadata_into_pdf.
|
||||||
|
|
||||||
# Check for file in multiple locations:
|
# Check for file in multiple locations:
|
||||||
# 1. Original location in tmp (file_record.local_filename)
|
# 1. Original location in tmp (file_record.local_filename)
|
||||||
# 2. Processed location (file_record.processed_file_path)
|
# 2. Processed location (file_record.processed_file_path)
|
||||||
@@ -536,21 +536,20 @@ def _retry_pipeline_step(file_record: FileRecord, step_name: str, db: Session) -
|
|||||||
elif file_record.processed_file_path and os.path.exists(file_record.processed_file_path):
|
elif file_record.processed_file_path and os.path.exists(file_record.processed_file_path):
|
||||||
# File has been processed and moved to processed directory
|
# File has been processed and moved to processed directory
|
||||||
file_path = file_record.processed_file_path
|
file_path = file_record.processed_file_path
|
||||||
else:
|
# Try fallback path in workdir/tmp
|
||||||
# Try fallback path in workdir/tmp
|
elif file_record.local_filename:
|
||||||
if file_record.local_filename:
|
workdir = settings.workdir
|
||||||
workdir = settings.workdir
|
tmp_dir = os.path.join(workdir, "tmp")
|
||||||
tmp_dir = os.path.join(workdir, "tmp")
|
fallback_path = os.path.join(tmp_dir, os.path.basename(file_record.local_filename))
|
||||||
fallback_path = os.path.join(tmp_dir, os.path.basename(file_record.local_filename))
|
if os.path.exists(fallback_path):
|
||||||
if os.path.exists(fallback_path):
|
file_path = fallback_path
|
||||||
file_path = fallback_path
|
|
||||||
|
|
||||||
if not file_path:
|
if not file_path:
|
||||||
raise HTTPException(
|
raise HTTPException(
|
||||||
status_code=400,
|
status_code=400,
|
||||||
detail="File not found in tmp or processed directory. Cannot retry metadata embedding."
|
detail="File not found in tmp or processed directory. Cannot retry metadata embedding."
|
||||||
)
|
)
|
||||||
|
|
||||||
extracted_text = _extract_text_from_pdf(file_path)
|
extracted_text = _extract_text_from_pdf(file_path)
|
||||||
# Pass the full path to the task so it can locate the file
|
# Pass the full path to the task so it can locate the file
|
||||||
task = extract_metadata_task.delay(file_path, extracted_text, file_id)
|
task = extract_metadata_task.delay(file_path, extracted_text, file_id)
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ def extract_json_from_text(text):
|
|||||||
def extract_metadata_with_gpt(self, filename: str, cleaned_text: str, file_id: int = None):
|
def extract_metadata_with_gpt(self, filename: str, cleaned_text: str, file_id: int = None):
|
||||||
"""
|
"""
|
||||||
Uses OpenAI to classify document metadata.
|
Uses OpenAI to classify document metadata.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
filename: Can be either a basename (e.g., "file.pdf") or a full path (e.g., "/workdir/processed/file.pdf")
|
filename: Can be either a basename (e.g., "file.pdf") or a full path (e.g., "/workdir/processed/file.pdf")
|
||||||
cleaned_text: The extracted text from the document
|
cleaned_text: The extracted text from the document
|
||||||
|
|||||||
Reference in New Issue
Block a user