fix(deduplication): handle UNIQUE constraint violation for duplicate files
- Remove duplicate record creation to avoid UNIQUE constraint on filehash - When duplicate detected, return original file_id instead of creating new record - Avoids sqlite3.IntegrityError: UNIQUE constraint failed - Simpler approach: duplicates not tracked as separate records, just rejected - Revert filehash column back to NOT NULL (required for original files) - Fixes error: (sqlite3.IntegrityError) UNIQUE constraint failed: files.filehash
This commit is contained in:
@@ -113,32 +113,19 @@ def process_document(self, original_local_file: str, original_filename: str = No
|
|||||||
existing = db.query(FileRecord).filter_by(filehash=filehash).one_or_none()
|
existing = db.query(FileRecord).filter_by(filehash=filehash).one_or_none()
|
||||||
if existing and settings.enable_deduplication:
|
if existing and settings.enable_deduplication:
|
||||||
logger.info(f"[{task_id}] Duplicate file detected (hash={filehash[:10]}...) Skipping processing.")
|
logger.info(f"[{task_id}] Duplicate file detected (hash={filehash[:10]}...) Skipping processing.")
|
||||||
# Create a file record for this duplicate with is_duplicate=True
|
# Log the deduplication result without creating a new database record
|
||||||
duplicate_record = FileRecord(
|
# This avoids UNIQUE constraint violations on filehash
|
||||||
filehash=filehash,
|
|
||||||
original_filename=original_filename,
|
|
||||||
local_filename="",
|
|
||||||
file_size=file_size,
|
|
||||||
mime_type=mime_type,
|
|
||||||
is_duplicate=True,
|
|
||||||
duplicate_of_id=existing.id,
|
|
||||||
)
|
|
||||||
db.add(duplicate_record)
|
|
||||||
db.commit()
|
|
||||||
db.refresh(duplicate_record)
|
|
||||||
|
|
||||||
if settings.enable_deduplication and settings.show_deduplication_step:
|
if settings.enable_deduplication and settings.show_deduplication_step:
|
||||||
log_task_progress(
|
log_task_progress(
|
||||||
task_id,
|
task_id,
|
||||||
"check_for_duplicates",
|
"check_for_duplicates",
|
||||||
"success",
|
"success",
|
||||||
f"Duplicate detected - matching file ID {existing.id}",
|
f"Duplicate detected - matching file ID {existing.id}",
|
||||||
file_id=duplicate_record.id,
|
file_id=existing.id,
|
||||||
detail=(
|
detail=(
|
||||||
f"Duplicate file detected.\n"
|
f"Duplicate file detected.\n"
|
||||||
f"File hash: {filehash}\n"
|
f"File hash: {filehash}\n"
|
||||||
f"Original file record ID: {existing.id}\n"
|
f"Original file record ID: {existing.id}\n"
|
||||||
f"This file record ID: {duplicate_record.id}\n"
|
|
||||||
f"Original filename: {original_filename}"
|
f"Original filename: {original_filename}"
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
@@ -147,7 +134,7 @@ def process_document(self, original_local_file: str, original_filename: str = No
|
|||||||
"process_document",
|
"process_document",
|
||||||
"success",
|
"success",
|
||||||
"Duplicate file detected, skipping",
|
"Duplicate file detected, skipping",
|
||||||
file_id=duplicate_record.id,
|
file_id=existing.id,
|
||||||
detail=(
|
detail=(
|
||||||
f"Duplicate file detected.\n"
|
f"Duplicate file detected.\n"
|
||||||
f"File hash: {filehash}\n"
|
f"File hash: {filehash}\n"
|
||||||
@@ -157,7 +144,7 @@ def process_document(self, original_local_file: str, original_filename: str = No
|
|||||||
)
|
)
|
||||||
return {
|
return {
|
||||||
"status": "duplicate_file",
|
"status": "duplicate_file",
|
||||||
"file_id": duplicate_record.id,
|
"file_id": existing.id,
|
||||||
"original_file_id": existing.id,
|
"original_file_id": existing.id,
|
||||||
"detail": "File already processed.",
|
"detail": "File already processed.",
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user