fix(config): default meilisearch_url to http://meilisearch:7700 for Docker/K8s service discovery
Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
@@ -299,3 +299,14 @@ NOTIFY_ON_FILE_PROCESSED=True
|
||||
# Uptime Kuma
|
||||
UPTIME_KUMA_URL=https://status.example.com/api/push/abcdef123456?status=up
|
||||
UPTIME_KUMA_PING_INTERVAL=5
|
||||
|
||||
# **Full-Text Search (Meilisearch)**
|
||||
# URL for the Meilisearch instance.
|
||||
# Default is "http://meilisearch:7700" — the Docker Compose / K8s service name —
|
||||
# so container-to-container networking works without extra configuration.
|
||||
# Override to "http://localhost:7700" only when running the API process outside Docker.
|
||||
MEILISEARCH_URL=http://meilisearch:7700
|
||||
# Optional master/API key for secured Meilisearch instances
|
||||
# MEILISEARCH_API_KEY=your_master_key_here
|
||||
MEILISEARCH_INDEX_NAME=documents
|
||||
ENABLE_SEARCH=True
|
||||
|
||||
@@ -15,6 +15,7 @@ from app.api.logs import router as logs_router
|
||||
from app.api.onedrive import router as onedrive_router
|
||||
from app.api.openai import router as openai_router
|
||||
from app.api.process import router as process_router
|
||||
from app.api.search import router as search_router
|
||||
from app.api.settings import router as settings_router
|
||||
from app.api.url_upload import router as url_upload_router
|
||||
|
||||
@@ -40,3 +41,4 @@ router.include_router(google_drive_router)
|
||||
router.include_router(logs_router)
|
||||
router.include_router(settings_router)
|
||||
router.include_router(url_upload_router)
|
||||
router.include_router(search_router)
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
"""Full-text search API endpoints.
|
||||
|
||||
Provides document search across OCR text, AI metadata, filenames, and tags
|
||||
via Meilisearch. Designed to serve as the backend for the UI search bar on
|
||||
the /files page and as a standalone API for integrations.
|
||||
|
||||
Future extension point: the OCR text stored in the index is also suitable
|
||||
for RAG (Retrieval Augmented Generation) chatbot workflows.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, Query, Request
|
||||
|
||||
from app.auth import require_login
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
@router.get("/search")
|
||||
@require_login
|
||||
def search_api(
|
||||
request: Request,
|
||||
q: str = Query(..., min_length=1, max_length=512, description="Full-text search query"),
|
||||
mime_type: Optional[str] = Query(None, description="Filter by MIME type (e.g. application/pdf)"),
|
||||
document_type: Optional[str] = Query(None, description="Filter by document type (e.g. Invoice)"),
|
||||
language: Optional[str] = Query(None, description="Filter by language code (e.g. de, en)"),
|
||||
date_from: Optional[int] = Query(None, description="Filter results created after this Unix timestamp"),
|
||||
date_to: Optional[int] = Query(None, description="Filter results created before this Unix timestamp"),
|
||||
page: int = Query(1, ge=1, description="Page number (1-based)"),
|
||||
per_page: int = Query(20, ge=1, le=100, description="Results per page"),
|
||||
):
|
||||
"""Search documents by full text, metadata, and tags.
|
||||
|
||||
Searches across:
|
||||
- Document title and filename
|
||||
- OCR / extracted text
|
||||
- Tags, sender, recipient, document type
|
||||
- Correspondent and reference number
|
||||
|
||||
Results are ranked by Meilisearch relevance and include highlighted
|
||||
snippets showing where the query terms matched.
|
||||
|
||||
Query Parameters:
|
||||
- q: Search query (required)
|
||||
- mime_type: Filter by MIME type
|
||||
- document_type: Filter by document type
|
||||
- language: Filter by language code
|
||||
- date_from: Unix timestamp lower bound
|
||||
- date_to: Unix timestamp upper bound
|
||||
- page: Page number (default: 1)
|
||||
- per_page: Results per page (default: 20, max: 100)
|
||||
|
||||
Example:
|
||||
```
|
||||
GET /api/search?q=invoice&document_type=Invoice&date_from=1704067200&page=1&per_page=20
|
||||
```
|
||||
|
||||
Response:
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"file_id": 42,
|
||||
"original_filename": "2026-01-15_Invoice_Amazon.pdf",
|
||||
"document_title": "Amazon Invoice January 2026",
|
||||
"document_type": "Invoice",
|
||||
"tags": ["amazon", "invoice"],
|
||||
"_formatted": {
|
||||
"document_title": "Amazon <mark>Invoice</mark> January 2026",
|
||||
"ocr_text": "...total amount of the <mark>invoice</mark> is..."
|
||||
}
|
||||
}
|
||||
],
|
||||
"total": 42,
|
||||
"page": 1,
|
||||
"pages": 3,
|
||||
"query": "invoice"
|
||||
}
|
||||
```
|
||||
"""
|
||||
logger.info(f"Search request: q={q!r}, mime_type={mime_type}, page={page}, per_page={per_page}")
|
||||
|
||||
result = search_documents(
|
||||
q,
|
||||
mime_type=mime_type,
|
||||
document_type=document_type,
|
||||
language=language,
|
||||
date_from=date_from,
|
||||
date_to=date_to,
|
||||
page=page,
|
||||
per_page=per_page,
|
||||
)
|
||||
|
||||
return result
|
||||
@@ -207,6 +207,15 @@ class Settings(BaseSettings):
|
||||
uptime_kuma_url: Optional[str] = None
|
||||
uptime_kuma_ping_interval: int = 5 # Default ping interval in minutes
|
||||
|
||||
# Meilisearch settings (full-text search engine)
|
||||
# Default uses the Docker Compose / K8s service name so container-to-container
|
||||
# networking works without any extra configuration. Override to
|
||||
# "http://localhost:7700" only when running the API process outside of Docker.
|
||||
meilisearch_url: str = "http://meilisearch:7700"
|
||||
meilisearch_api_key: Optional[str] = None # Master or API key (optional for local dev)
|
||||
meilisearch_index_name: str = "documents"
|
||||
enable_search: bool = True # Enable Meilisearch full-text search integration
|
||||
|
||||
# HTTP request settings
|
||||
http_request_timeout: int = 120 # Default timeout for HTTP requests in seconds (handles large file operations)
|
||||
|
||||
|
||||
@@ -55,6 +55,15 @@ class FileRecord(Base):
|
||||
# If this is a duplicate, record the ID of the original file for reference
|
||||
duplicate_of_id = Column(Integer, ForeignKey(_FILES_ID_FK), nullable=True)
|
||||
|
||||
# Full OCR/extracted text for full-text search and RAG
|
||||
ocr_text = Column(Text, nullable=True)
|
||||
|
||||
# AI-extracted metadata stored as JSON string (filename, tags, title, sender, etc.)
|
||||
ai_metadata = Column(Text, nullable=True)
|
||||
|
||||
# Human-readable document title from AI metadata
|
||||
document_title = Column(String, nullable=True)
|
||||
|
||||
# Timestamp when we inserted this record
|
||||
created_at = Column(DateTime(timezone=True), server_default=func.now())
|
||||
|
||||
|
||||
@@ -195,8 +195,26 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met
|
||||
original_file_path = file_record.original_file_path
|
||||
# Update the processed_file_path in the database
|
||||
file_record.processed_file_path = final_file_path
|
||||
# Persist extracted text and AI metadata to DB for full-text search / RAG
|
||||
file_record.ocr_text = extracted_text or None
|
||||
if metadata:
|
||||
try:
|
||||
file_record.ai_metadata = json.dumps(metadata, ensure_ascii=False)
|
||||
except Exception:
|
||||
pass
|
||||
file_record.document_title = (
|
||||
metadata.get("title") or metadata.get("filename") or file_record.original_filename
|
||||
)
|
||||
db.commit()
|
||||
logger.info(f"[{task_id}] Updated database with processed_file_path: {final_file_path}")
|
||||
logger.info(f"[{task_id}] Updated database with processed_file_path and search fields")
|
||||
|
||||
# Index into Meilisearch for full-text search (non-blocking, best-effort)
|
||||
try:
|
||||
from app.utils.meilisearch_client import index_document
|
||||
|
||||
index_document(file_record, extracted_text or "", metadata or {})
|
||||
except Exception as search_exc:
|
||||
logger.warning(f"[{task_id}] Meilisearch indexing failed (non-fatal): {search_exc}")
|
||||
|
||||
# Persist the metadata into a JSON file with the same base name.
|
||||
# Include file path references for traceability
|
||||
|
||||
@@ -0,0 +1,290 @@
|
||||
"""Meilisearch client utilities for full-text document search.
|
||||
|
||||
This module provides functions for indexing documents into Meilisearch
|
||||
and searching across OCR text, AI metadata, filenames, and tags.
|
||||
|
||||
The search index is designed to support future RAG (Retrieval Augmented
|
||||
Generation) workflows by storing full document text alongside structured
|
||||
metadata fields.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Index settings applied once at index creation / first use
|
||||
_INDEX_SETTINGS = {
|
||||
"searchableAttributes": [
|
||||
"document_title",
|
||||
"original_filename",
|
||||
"ocr_text",
|
||||
"tags",
|
||||
"sender",
|
||||
"recipient",
|
||||
"document_type",
|
||||
"correspondent",
|
||||
],
|
||||
"filterableAttributes": [
|
||||
"mime_type",
|
||||
"document_type",
|
||||
"language",
|
||||
"tags",
|
||||
"created_at_ts",
|
||||
"file_id",
|
||||
],
|
||||
"sortableAttributes": [
|
||||
"created_at_ts",
|
||||
"file_size",
|
||||
],
|
||||
"displayedAttributes": [
|
||||
"file_id",
|
||||
"original_filename",
|
||||
"document_title",
|
||||
"document_type",
|
||||
"tags",
|
||||
"sender",
|
||||
"recipient",
|
||||
"correspondent",
|
||||
"language",
|
||||
"reference_number",
|
||||
"mime_type",
|
||||
"file_size",
|
||||
"created_at_ts",
|
||||
"ocr_text",
|
||||
],
|
||||
"rankingRules": [
|
||||
"words",
|
||||
"typo",
|
||||
"proximity",
|
||||
"attribute",
|
||||
"sort",
|
||||
"exactness",
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def get_meilisearch_client():
|
||||
"""Return a configured Meilisearch client, or None if unavailable/disabled."""
|
||||
try:
|
||||
import meilisearch
|
||||
|
||||
from app.config import settings
|
||||
|
||||
if not settings.enable_search:
|
||||
return None
|
||||
|
||||
kwargs: dict[str, Any] = {"url": settings.meilisearch_url}
|
||||
if settings.meilisearch_api_key:
|
||||
kwargs["api_key"] = settings.meilisearch_api_key
|
||||
|
||||
client = meilisearch.Client(**kwargs)
|
||||
return client
|
||||
except ImportError:
|
||||
logger.warning("meilisearch package not installed; search disabled")
|
||||
return None
|
||||
except Exception as exc:
|
||||
logger.warning(f"Could not connect to Meilisearch: {exc}")
|
||||
return None
|
||||
|
||||
|
||||
def _get_or_create_index(client):
|
||||
"""Get the documents index, creating it with settings if it doesn't exist."""
|
||||
from app.config import settings
|
||||
|
||||
index_name = settings.meilisearch_index_name
|
||||
try:
|
||||
index = client.get_index(index_name)
|
||||
except Exception:
|
||||
# Index doesn't exist – create it with file_id as primary key
|
||||
task = client.create_index(index_name, {"primaryKey": "file_id"})
|
||||
client.wait_for_task(task.task_uid)
|
||||
index = client.get_index(index_name)
|
||||
# Apply search settings
|
||||
try:
|
||||
task = index.update_settings(_INDEX_SETTINGS)
|
||||
client.wait_for_task(task.task_uid)
|
||||
except Exception as exc:
|
||||
logger.warning(f"Could not update Meilisearch index settings: {exc}")
|
||||
return index
|
||||
|
||||
|
||||
def _build_document(file_record, text: str, metadata: dict) -> dict:
|
||||
"""Build a Meilisearch document from a FileRecord and extracted content."""
|
||||
import json as _json
|
||||
|
||||
tags = metadata.get("tags", [])
|
||||
if isinstance(tags, str):
|
||||
tags = [t.strip() for t in tags.split(",") if t.strip()]
|
||||
|
||||
# Unix timestamp for sorting/filtering
|
||||
created_at_ts = 0
|
||||
if file_record.created_at:
|
||||
try:
|
||||
created_at_ts = int(file_record.created_at.timestamp())
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return {
|
||||
"file_id": file_record.id,
|
||||
"original_filename": file_record.original_filename or "",
|
||||
"document_title": metadata.get("title") or metadata.get("filename") or file_record.original_filename or "",
|
||||
"document_type": metadata.get("document_type") or metadata.get("kommunikationsart") or "",
|
||||
"tags": tags,
|
||||
"sender": metadata.get("absender") or "",
|
||||
"recipient": metadata.get("empfaenger") or "",
|
||||
"correspondent": metadata.get("correspondent") or "",
|
||||
"language": metadata.get("language") or "",
|
||||
"reference_number": metadata.get("reference_number") or "",
|
||||
"mime_type": file_record.mime_type or "",
|
||||
"file_size": file_record.file_size or 0,
|
||||
"created_at_ts": created_at_ts,
|
||||
"ocr_text": text or "",
|
||||
}
|
||||
|
||||
|
||||
def index_document(file_record, text: str, metadata: dict) -> bool:
|
||||
"""Index a document in Meilisearch.
|
||||
|
||||
Args:
|
||||
file_record: FileRecord ORM instance with at minimum .id set.
|
||||
text: Full OCR / extracted text for the document.
|
||||
metadata: AI-extracted metadata dict.
|
||||
|
||||
Returns:
|
||||
True if indexing succeeded, False otherwise.
|
||||
"""
|
||||
client = get_meilisearch_client()
|
||||
if client is None:
|
||||
return False
|
||||
|
||||
try:
|
||||
index = _get_or_create_index(client)
|
||||
doc = _build_document(file_record, text, metadata)
|
||||
task = index.add_documents([doc])
|
||||
logger.info(f"Queued Meilisearch indexing for file_id={file_record.id} (task_uid={task.task_uid})")
|
||||
return True
|
||||
except Exception as exc:
|
||||
logger.warning(f"Meilisearch indexing failed for file_id={file_record.id}: {exc}")
|
||||
return False
|
||||
|
||||
|
||||
def delete_document(file_id: int) -> bool:
|
||||
"""Remove a document from the Meilisearch index.
|
||||
|
||||
Args:
|
||||
file_id: The database ID of the file to remove.
|
||||
|
||||
Returns:
|
||||
True if deletion succeeded, False otherwise.
|
||||
"""
|
||||
client = get_meilisearch_client()
|
||||
if client is None:
|
||||
return False
|
||||
|
||||
try:
|
||||
from app.config import settings
|
||||
|
||||
index = client.get_index(settings.meilisearch_index_name)
|
||||
task = index.delete_document(file_id)
|
||||
logger.info(f"Queued Meilisearch deletion for file_id={file_id} (task_uid={task.task_uid})")
|
||||
return True
|
||||
except Exception as exc:
|
||||
logger.warning(f"Meilisearch deletion failed for file_id={file_id}: {exc}")
|
||||
return False
|
||||
|
||||
|
||||
def search_documents(
|
||||
query: str,
|
||||
*,
|
||||
mime_type: Optional[str] = None,
|
||||
document_type: Optional[str] = None,
|
||||
language: Optional[str] = None,
|
||||
date_from: Optional[int] = None,
|
||||
date_to: Optional[int] = None,
|
||||
page: int = 1,
|
||||
per_page: int = 20,
|
||||
) -> dict:
|
||||
"""Search documents in Meilisearch.
|
||||
|
||||
Args:
|
||||
query: Full-text search query string.
|
||||
mime_type: Optional MIME-type filter.
|
||||
document_type: Optional document type filter.
|
||||
language: Optional language filter (ISO 639-1, e.g. "de").
|
||||
date_from: Optional lower bound Unix timestamp for created_at.
|
||||
date_to: Optional upper bound Unix timestamp for created_at.
|
||||
page: 1-based page number.
|
||||
per_page: Results per page (max 100).
|
||||
|
||||
Returns:
|
||||
Dict with keys: results, total, page, pages, query.
|
||||
Returns empty results dict on any error.
|
||||
"""
|
||||
empty: dict = {"results": [], "total": 0, "page": page, "pages": 0, "query": query}
|
||||
|
||||
client = get_meilisearch_client()
|
||||
if client is None:
|
||||
return empty
|
||||
|
||||
try:
|
||||
from app.config import settings
|
||||
|
||||
index = _get_or_create_index(client)
|
||||
|
||||
# Build filter expressions
|
||||
filters: list[str] = []
|
||||
if mime_type:
|
||||
filters.append(f'mime_type = "{mime_type}"')
|
||||
if document_type:
|
||||
filters.append(f'document_type = "{document_type}"')
|
||||
if language:
|
||||
filters.append(f'language = "{language}"')
|
||||
if date_from is not None:
|
||||
filters.append(f"created_at_ts >= {date_from}")
|
||||
if date_to is not None:
|
||||
filters.append(f"created_at_ts <= {date_to}")
|
||||
|
||||
search_params: dict[str, Any] = {
|
||||
"offset": (page - 1) * per_page,
|
||||
"limit": per_page,
|
||||
"attributesToHighlight": ["document_title", "original_filename", "ocr_text", "tags"],
|
||||
"highlightPreTag": "<mark>",
|
||||
"highlightPostTag": "</mark>",
|
||||
"attributesToCrop": ["ocr_text"],
|
||||
"cropLength": 200,
|
||||
}
|
||||
|
||||
if filters:
|
||||
search_params["filter"] = " AND ".join(filters)
|
||||
|
||||
result = index.search(query, search_params)
|
||||
|
||||
hits = result.get("hits", [])
|
||||
total = result.get("estimatedTotalHits", result.get("nbHits", len(hits)))
|
||||
|
||||
# Attach highlights to each hit
|
||||
formatted_results = []
|
||||
for hit in hits:
|
||||
formatted = dict(hit)
|
||||
# Include formatted (highlighted) snippets if available
|
||||
if "_formatted" in hit:
|
||||
formatted["_formatted"] = hit["_formatted"]
|
||||
# Exclude raw ocr_text from results (use _formatted snippet instead)
|
||||
formatted.pop("ocr_text", None)
|
||||
formatted_results.append(formatted)
|
||||
|
||||
pages = (total + per_page - 1) // per_page if total > 0 else 0
|
||||
|
||||
return {
|
||||
"results": formatted_results,
|
||||
"total": total,
|
||||
"page": page,
|
||||
"pages": pages,
|
||||
"query": query,
|
||||
}
|
||||
|
||||
except Exception as exc:
|
||||
logger.warning(f"Meilisearch search failed for query '{query}': {exc}")
|
||||
return empty
|
||||
@@ -59,6 +59,15 @@ services:
|
||||
container_name: gotenberg
|
||||
restart: always
|
||||
|
||||
meilisearch:
|
||||
image: getmeili/meilisearch:latest
|
||||
container_name: document_meilisearch
|
||||
restart: always
|
||||
environment:
|
||||
- MEILI_NO_ANALYTICS=true
|
||||
volumes:
|
||||
- /var/docparse/meilisearch:/meili_data
|
||||
|
||||
redis:
|
||||
image: redis:alpine
|
||||
container_name: document_redis
|
||||
|
||||
@@ -418,6 +418,47 @@
|
||||
</form>
|
||||
</div>
|
||||
|
||||
<!-- Full-Text Search Section -->
|
||||
<div class="filters-section" style="margin-top: 0.75rem;">
|
||||
<div style="width: 100%;">
|
||||
<label for="fulltext-search" style="font-weight: 600; display: block; margin-bottom: 0.4rem;">
|
||||
<i class="fas fa-search"></i> Full-Text Search (OCR text, metadata, tags)
|
||||
</label>
|
||||
<div style="display: flex; gap: 0.5rem; align-items: center; flex-wrap: wrap;">
|
||||
<input
|
||||
type="text"
|
||||
id="fulltext-search"
|
||||
placeholder="Search document content, sender, tags, type..."
|
||||
style="flex: 1; min-width: 220px; padding: 0.5rem 0.75rem; border: 1px solid #d1d5db; border-radius: 0.375rem; font-size: 0.875rem;"
|
||||
oninput="debounceSearch(this.value)"
|
||||
>
|
||||
<button
|
||||
type="button"
|
||||
onclick="runFullTextSearch()"
|
||||
style="padding: 0.5rem 1rem; background-color: #3182ce; color: white; border: none; border-radius: 0.375rem; cursor: pointer; font-size: 0.875rem; white-space: nowrap;"
|
||||
>
|
||||
Search
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
onclick="clearFullTextSearch()"
|
||||
style="padding: 0.5rem 1rem; background-color: #6b7280; color: white; border: none; border-radius: 0.375rem; cursor: pointer; font-size: 0.875rem; white-space: nowrap;"
|
||||
>
|
||||
Clear
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Full-Text Search Results Panel -->
|
||||
<div id="search-results-panel" style="display: none; margin-top: 0.5rem; border: 1px solid #e5e7eb; border-radius: 0.5rem; background: white; box-shadow: 0 1px 3px rgba(0,0,0,0.08);">
|
||||
<div style="padding: 0.75rem 1rem; border-bottom: 1px solid #e5e7eb; display: flex; justify-content: space-between; align-items: center; background: #f9fafb; border-radius: 0.5rem 0.5rem 0 0;">
|
||||
<span id="search-results-summary" style="font-size: 0.875rem; color: #374151;"></span>
|
||||
<div id="search-results-pagination" style="display: flex; gap: 0.5rem;"></div>
|
||||
</div>
|
||||
<div id="search-results-list" style="padding: 0.5rem 0;"></div>
|
||||
</div>
|
||||
|
||||
<!-- Bulk Actions Section -->
|
||||
<div id="bulkActionsBar" class="filters-section" style="display: none; background-color: #e6f3ff;">
|
||||
<div class="flex flex-col sm:flex-row sm:justify-between sm:items-center gap-3">
|
||||
@@ -887,6 +928,117 @@
|
||||
function closeUploadModal() {
|
||||
uploadModal.classList.remove('active');
|
||||
}
|
||||
|
||||
// ---- Full-Text Search (Meilisearch) ----
|
||||
let _searchDebounceTimer = null;
|
||||
let _searchCurrentPage = 1;
|
||||
|
||||
function debounceSearch(value) {
|
||||
clearTimeout(_searchDebounceTimer);
|
||||
if (!value || value.trim().length < 2) {
|
||||
clearFullTextSearch();
|
||||
return;
|
||||
}
|
||||
_searchDebounceTimer = setTimeout(() => {
|
||||
_searchCurrentPage = 1;
|
||||
runFullTextSearch();
|
||||
}, 400);
|
||||
}
|
||||
|
||||
function runFullTextSearch(page) {
|
||||
const input = document.getElementById('fulltext-search');
|
||||
const q = input ? input.value.trim() : '';
|
||||
if (!q) { clearFullTextSearch(); return; }
|
||||
if (page) _searchCurrentPage = page;
|
||||
|
||||
const panel = document.getElementById('search-results-panel');
|
||||
const list = document.getElementById('search-results-list');
|
||||
const summary = document.getElementById('search-results-summary');
|
||||
const pagination = document.getElementById('search-results-pagination');
|
||||
|
||||
list.innerHTML = '<div style="padding: 1rem; color: #6b7280; font-size: 0.875rem;"><i class="fas fa-spinner fa-spin"></i> Searching…</div>';
|
||||
summary.textContent = '';
|
||||
pagination.innerHTML = '';
|
||||
panel.style.display = 'block';
|
||||
|
||||
const params = new URLSearchParams({ q, page: _searchCurrentPage, per_page: 20 });
|
||||
fetch(`/api/search?${params.toString()}`)
|
||||
.then(r => {
|
||||
if (!r.ok) throw new Error(`Search returned ${r.status}`);
|
||||
return r.json();
|
||||
})
|
||||
.then(data => renderSearchResults(data, q))
|
||||
.catch(err => {
|
||||
list.innerHTML = `<div style="padding: 1rem; color: #dc2626; font-size: 0.875rem;"><i class="fas fa-exclamation-triangle"></i> Search unavailable: ${err.message}</div>`;
|
||||
summary.textContent = '';
|
||||
});
|
||||
}
|
||||
|
||||
function renderSearchResults(data, q) {
|
||||
const panel = document.getElementById('search-results-panel');
|
||||
const list = document.getElementById('search-results-list');
|
||||
const summary = document.getElementById('search-results-summary');
|
||||
const pagination = document.getElementById('search-results-pagination');
|
||||
|
||||
const { results, total, page, pages } = data;
|
||||
summary.textContent = `${total} result${total !== 1 ? 's' : ''} for "${q}"`;
|
||||
|
||||
if (!results || results.length === 0) {
|
||||
list.innerHTML = '<div style="padding: 1rem; color: #6b7280; font-size: 0.875rem;">No results found.</div>';
|
||||
pagination.innerHTML = '';
|
||||
return;
|
||||
}
|
||||
|
||||
list.innerHTML = results.map(hit => {
|
||||
const fmt = hit._formatted || {};
|
||||
const title = fmt.document_title || hit.document_title || hit.original_filename || '(untitled)';
|
||||
const filename = fmt.original_filename || hit.original_filename || '';
|
||||
const snippet = fmt.ocr_text || '';
|
||||
const tags = Array.isArray(hit.tags) ? hit.tags.join(', ') : (hit.tags || '');
|
||||
const docType = hit.document_type || '';
|
||||
|
||||
return `<div style="padding: 0.75rem 1rem; border-bottom: 1px solid #f3f4f6; display: flex; gap: 0.75rem; align-items: flex-start;">
|
||||
<div style="flex-shrink: 0; color: #3b82f6; font-size: 1.25rem; padding-top: 0.1rem;">
|
||||
<i class="fas fa-file-pdf"></i>
|
||||
</div>
|
||||
<div style="flex: 1; min-width: 0;">
|
||||
<div style="font-weight: 600; font-size: 0.9rem; color: #111827;">${title}</div>
|
||||
${filename ? `<div style="font-size: 0.8rem; color: #6b7280; margin-top: 0.15rem;">${filename}</div>` : ''}
|
||||
${docType ? `<span style="display: inline-block; margin-top: 0.25rem; padding: 0.1rem 0.5rem; background: #eff6ff; color: #1d4ed8; border-radius: 9999px; font-size: 0.75rem;">${docType}</span>` : ''}
|
||||
${tags ? `<span style="display: inline-block; margin-top: 0.25rem; margin-left: 0.25rem; padding: 0.1rem 0.5rem; background: #f0fdf4; color: #15803d; border-radius: 9999px; font-size: 0.75rem;">${tags}</span>` : ''}
|
||||
${snippet ? `<div style="margin-top: 0.4rem; font-size: 0.8rem; color: #374151; white-space: pre-wrap; word-break: break-word;">…${snippet}…</div>` : ''}
|
||||
</div>
|
||||
<div style="flex-shrink: 0;">
|
||||
<a href="/files/${hit.file_id}" style="padding: 0.25rem 0.6rem; background: #f3f4f6; color: #374151; border-radius: 0.25rem; font-size: 0.8rem; text-decoration: none; white-space: nowrap;" title="View file">
|
||||
<i class="fas fa-external-link-alt"></i>
|
||||
</a>
|
||||
</div>
|
||||
</div>`;
|
||||
}).join('');
|
||||
|
||||
// Pagination
|
||||
if (pages > 1) {
|
||||
const btns = [];
|
||||
if (page > 1) {
|
||||
btns.push(`<button onclick="runFullTextSearch(${page - 1})" style="padding: 0.25rem 0.6rem; border: 1px solid #d1d5db; border-radius: 0.25rem; font-size: 0.8rem; cursor: pointer; background: white;">« Prev</button>`);
|
||||
}
|
||||
btns.push(`<span style="padding: 0.25rem 0.6rem; font-size: 0.8rem; color: #6b7280;">Page ${page} / ${pages}</span>`);
|
||||
if (page < pages) {
|
||||
btns.push(`<button onclick="runFullTextSearch(${page + 1})" style="padding: 0.25rem 0.6rem; border: 1px solid #d1d5db; border-radius: 0.25rem; font-size: 0.8rem; cursor: pointer; background: white;">Next »</button>`);
|
||||
}
|
||||
pagination.innerHTML = btns.join('');
|
||||
} else {
|
||||
pagination.innerHTML = '';
|
||||
}
|
||||
}
|
||||
|
||||
function clearFullTextSearch() {
|
||||
const panel = document.getElementById('search-results-panel');
|
||||
const input = document.getElementById('fulltext-search');
|
||||
if (panel) panel.style.display = 'none';
|
||||
if (input) input.value = '';
|
||||
_searchCurrentPage = 1;
|
||||
}
|
||||
</script>
|
||||
</div>
|
||||
{% endblock %}
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Add OCR text and AI metadata fields to files table for full-text search
|
||||
|
||||
Revision ID: 004_add_search_fields
|
||||
Revises: 003_add_deduplication_support
|
||||
Create Date: 2026-02-25
|
||||
|
||||
"""
|
||||
|
||||
from typing import Union
|
||||
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision: str = "004_add_search_fields"
|
||||
down_revision: Union[str, None] = "003_add_deduplication_support"
|
||||
depends_on: Union[str, None] = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
"""Add ocr_text, ai_metadata, and document_title columns to files table."""
|
||||
op.add_column("files", sa.Column("ocr_text", sa.Text(), nullable=True))
|
||||
op.add_column("files", sa.Column("ai_metadata", sa.Text(), nullable=True))
|
||||
op.add_column("files", sa.Column("document_title", sa.String(), nullable=True))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
"""Remove ocr_text, ai_metadata, and document_title columns from files table."""
|
||||
op.drop_column("files", "document_title")
|
||||
op.drop_column("files", "ai_metadata")
|
||||
op.drop_column("files", "ocr_text")
|
||||
@@ -43,3 +43,4 @@ litellm>=1.0.0,<2.0.0
|
||||
pytesseract>=0.3.10 # Python wrapper for Tesseract OCR
|
||||
pdf2image>=1.17.0 # Convert PDF pages to images (used by Tesseract and EasyOCR providers)
|
||||
ocrmypdf>=16.0.0,<17.0.0 # Post-processing: embeds searchable text layers into PDFs via Tesseract
|
||||
meilisearch>=0.31.0 # Full-text search engine client
|
||||
|
||||
@@ -0,0 +1,313 @@
|
||||
"""Tests for the full-text search API (app/api/search.py) and Meilisearch
|
||||
client utilities (app/utils/meilisearch_client.py).
|
||||
"""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# app/utils/meilisearch_client tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestMeilisearchClientDisabled:
|
||||
"""Tests when search is disabled or Meilisearch is unavailable."""
|
||||
|
||||
def test_get_client_returns_none_when_disabled(self):
|
||||
"""get_meilisearch_client returns None when enable_search=False."""
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=None):
|
||||
from app.utils.meilisearch_client import index_document, search_documents
|
||||
|
||||
class _FakeRecord:
|
||||
id = 1
|
||||
original_filename = "test.pdf"
|
||||
mime_type = "application/pdf"
|
||||
file_size = 1024
|
||||
created_at = None
|
||||
|
||||
result = index_document(_FakeRecord(), "some text", {})
|
||||
assert result is False
|
||||
|
||||
result = search_documents("invoice")
|
||||
assert result["results"] == []
|
||||
assert result["total"] == 0
|
||||
|
||||
def test_search_documents_import_error(self):
|
||||
"""search_documents returns empty dict when meilisearch not installed."""
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=None):
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
result = search_documents("test")
|
||||
assert result["results"] == []
|
||||
assert result["total"] == 0
|
||||
assert result["page"] == 1
|
||||
assert result["query"] == "test"
|
||||
|
||||
def test_delete_document_no_client(self):
|
||||
"""delete_document returns False when client unavailable."""
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=None):
|
||||
from app.utils.meilisearch_client import delete_document
|
||||
|
||||
result = delete_document(99)
|
||||
assert result is False
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestMeilisearchIndexDocument:
|
||||
"""Tests for index_document function."""
|
||||
|
||||
def test_index_document_success(self):
|
||||
"""index_document returns True when Meilisearch succeeds."""
|
||||
mock_client = MagicMock()
|
||||
mock_index = MagicMock()
|
||||
mock_task = MagicMock()
|
||||
mock_task.task_uid = 1
|
||||
|
||||
mock_client.get_index.return_value = mock_index
|
||||
mock_index.add_documents.return_value = mock_task
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import index_document
|
||||
|
||||
class _FakeRecord:
|
||||
id = 1
|
||||
original_filename = "invoice.pdf"
|
||||
mime_type = "application/pdf"
|
||||
file_size = 2048
|
||||
created_at = None
|
||||
|
||||
result = index_document(
|
||||
_FakeRecord(),
|
||||
"This is an invoice for services rendered",
|
||||
{
|
||||
"title": "Invoice January 2026",
|
||||
"document_type": "Invoice",
|
||||
"tags": ["invoice", "services"],
|
||||
"absender": "ACME Corp",
|
||||
"language": "en",
|
||||
},
|
||||
)
|
||||
assert result is True
|
||||
mock_index.add_documents.assert_called_once()
|
||||
call_docs = mock_index.add_documents.call_args[0][0]
|
||||
assert len(call_docs) == 1
|
||||
doc = call_docs[0]
|
||||
assert doc["file_id"] == 1
|
||||
assert doc["document_title"] == "Invoice January 2026"
|
||||
assert "invoice" in doc["tags"]
|
||||
|
||||
def test_index_document_meilisearch_error(self):
|
||||
"""index_document returns False on Meilisearch exception."""
|
||||
mock_client = MagicMock()
|
||||
mock_index = MagicMock()
|
||||
mock_client.get_index.return_value = mock_index
|
||||
mock_index.add_documents.side_effect = RuntimeError("Meilisearch down")
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import index_document
|
||||
|
||||
class _FakeRecord:
|
||||
id = 2
|
||||
original_filename = "test.pdf"
|
||||
mime_type = "application/pdf"
|
||||
file_size = 512
|
||||
created_at = None
|
||||
|
||||
result = index_document(_FakeRecord(), "some text", {})
|
||||
assert result is False
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestMeilisearchSearchDocuments:
|
||||
"""Tests for search_documents function."""
|
||||
|
||||
def _make_mock_client(self, hits=None, total=None):
|
||||
mock_client = MagicMock()
|
||||
mock_index = MagicMock()
|
||||
mock_client.get_index.return_value = mock_index
|
||||
mock_index.search.return_value = {
|
||||
"hits": hits or [],
|
||||
"estimatedTotalHits": total if total is not None else len(hits or []),
|
||||
}
|
||||
return mock_client, mock_index
|
||||
|
||||
def test_search_returns_results(self):
|
||||
"""search_documents returns hits from Meilisearch."""
|
||||
hits = [
|
||||
{
|
||||
"file_id": 42,
|
||||
"original_filename": "2026-01-15_Invoice_Amazon.pdf",
|
||||
"document_title": "Amazon Invoice",
|
||||
"document_type": "Invoice",
|
||||
"tags": ["amazon", "invoice"],
|
||||
"ocr_text": "Amazon invoice content here",
|
||||
"_formatted": {
|
||||
"document_title": "Amazon <mark>Invoice</mark>",
|
||||
"ocr_text": "…Amazon <mark>invoice</mark> content here…",
|
||||
},
|
||||
}
|
||||
]
|
||||
mock_client, mock_index = self._make_mock_client(hits=hits, total=1)
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
result = search_documents("invoice", page=1, per_page=20)
|
||||
|
||||
assert result["total"] == 1
|
||||
assert result["pages"] == 1
|
||||
assert result["query"] == "invoice"
|
||||
assert len(result["results"]) == 1
|
||||
# Raw ocr_text should be stripped from result (only _formatted snippet kept)
|
||||
assert "ocr_text" not in result["results"][0]
|
||||
assert result["results"][0]["file_id"] == 42
|
||||
|
||||
def test_search_with_filters(self):
|
||||
"""search_documents passes filter expressions to Meilisearch."""
|
||||
mock_client, mock_index = self._make_mock_client(hits=[], total=0)
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
search_documents("contract", mime_type="application/pdf", language="de", page=1, per_page=10)
|
||||
|
||||
call_kwargs = mock_index.search.call_args
|
||||
search_params = call_kwargs[0][1]
|
||||
assert "filter" in search_params
|
||||
assert 'mime_type = "application/pdf"' in search_params["filter"]
|
||||
assert 'language = "de"' in search_params["filter"]
|
||||
|
||||
def test_search_pagination(self):
|
||||
"""search_documents applies correct offset for page 2."""
|
||||
mock_client, mock_index = self._make_mock_client(hits=[], total=50)
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
result = search_documents("test", page=3, per_page=10)
|
||||
|
||||
call_kwargs = mock_index.search.call_args
|
||||
search_params = call_kwargs[0][1]
|
||||
assert search_params["offset"] == 20 # (3-1) * 10
|
||||
assert search_params["limit"] == 10
|
||||
assert result["pages"] == 5
|
||||
|
||||
def test_search_empty_results(self):
|
||||
"""search_documents returns proper empty structure."""
|
||||
mock_client, mock_index = self._make_mock_client(hits=[], total=0)
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
result = search_documents("nothing")
|
||||
|
||||
assert result["results"] == []
|
||||
assert result["total"] == 0
|
||||
assert result["pages"] == 0
|
||||
|
||||
def test_search_exception_returns_empty(self):
|
||||
"""search_documents returns empty dict on Meilisearch exception."""
|
||||
mock_client = MagicMock()
|
||||
mock_index = MagicMock()
|
||||
mock_client.get_index.side_effect = RuntimeError("Meilisearch unavailable")
|
||||
|
||||
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
|
||||
from app.utils.meilisearch_client import search_documents
|
||||
|
||||
result = search_documents("invoice")
|
||||
|
||||
assert result["results"] == []
|
||||
assert result["total"] == 0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Search API endpoint tests (GET /api/search)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestSearchAPIEndpoint:
|
||||
"""Tests for GET /api/search endpoint."""
|
||||
|
||||
def test_search_endpoint_success(self, client):
|
||||
"""GET /api/search?q=... returns search results."""
|
||||
mock_result = {
|
||||
"results": [{"file_id": 1, "document_title": "Test Invoice", "document_type": "Invoice"}],
|
||||
"total": 1,
|
||||
"page": 1,
|
||||
"pages": 1,
|
||||
"query": "invoice",
|
||||
}
|
||||
with patch("app.api.search.search_documents", return_value=mock_result):
|
||||
response = client.get("/api/search?q=invoice")
|
||||
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert data["total"] == 1
|
||||
assert data["query"] == "invoice"
|
||||
assert len(data["results"]) == 1
|
||||
|
||||
def test_search_endpoint_missing_query(self, client):
|
||||
"""GET /api/search without q returns 422."""
|
||||
response = client.get("/api/search")
|
||||
assert response.status_code == 422
|
||||
|
||||
def test_search_endpoint_empty_query(self, client):
|
||||
"""GET /api/search?q= (empty) returns 422 due to min_length=1."""
|
||||
response = client.get("/api/search?q=")
|
||||
assert response.status_code == 422
|
||||
|
||||
def test_search_endpoint_with_filters(self, client):
|
||||
"""GET /api/search with optional filters passes them to search_documents."""
|
||||
mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "invoice"}
|
||||
with patch("app.api.search.search_documents", return_value=mock_result) as mock_search:
|
||||
response = client.get("/api/search?q=invoice&mime_type=application/pdf&language=en&page=2&per_page=10")
|
||||
|
||||
assert response.status_code == 200
|
||||
mock_search.assert_called_once_with(
|
||||
"invoice",
|
||||
mime_type="application/pdf",
|
||||
document_type=None,
|
||||
language="en",
|
||||
date_from=None,
|
||||
date_to=None,
|
||||
page=2,
|
||||
per_page=10,
|
||||
)
|
||||
|
||||
def test_search_endpoint_per_page_max(self, client):
|
||||
"""GET /api/search with per_page > 100 returns 422."""
|
||||
response = client.get("/api/search?q=test&per_page=200")
|
||||
assert response.status_code == 422
|
||||
|
||||
def test_search_endpoint_pagination_defaults(self, client):
|
||||
"""GET /api/search uses default page=1 per_page=20."""
|
||||
mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "test"}
|
||||
with patch("app.api.search.search_documents", return_value=mock_result) as mock_search:
|
||||
response = client.get("/api/search?q=test")
|
||||
|
||||
assert response.status_code == 200
|
||||
mock_search.assert_called_once_with(
|
||||
"test",
|
||||
mime_type=None,
|
||||
document_type=None,
|
||||
language=None,
|
||||
date_from=None,
|
||||
date_to=None,
|
||||
page=1,
|
||||
per_page=20,
|
||||
)
|
||||
|
||||
def test_search_endpoint_date_filters(self, client):
|
||||
"""GET /api/search with date_from and date_to passes them as int."""
|
||||
mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "contract"}
|
||||
with patch("app.api.search.search_documents", return_value=mock_result) as mock_search:
|
||||
response = client.get("/api/search?q=contract&date_from=1704067200&date_to=1735689600")
|
||||
|
||||
assert response.status_code == 200
|
||||
call_kwargs = mock_search.call_args
|
||||
assert call_kwargs[1]["date_from"] == 1704067200
|
||||
assert call_kwargs[1]["date_to"] == 1735689600
|
||||
Reference in New Issue
Block a user