diff --git a/.env.demo b/.env.demo index 12f9ebf8..a54995f2 100644 --- a/.env.demo +++ b/.env.demo @@ -299,3 +299,14 @@ NOTIFY_ON_FILE_PROCESSED=True # Uptime Kuma UPTIME_KUMA_URL=https://status.example.com/api/push/abcdef123456?status=up UPTIME_KUMA_PING_INTERVAL=5 + +# **Full-Text Search (Meilisearch)** +# URL for the Meilisearch instance. +# Default is "http://meilisearch:7700" — the Docker Compose / K8s service name — +# so container-to-container networking works without extra configuration. +# Override to "http://localhost:7700" only when running the API process outside Docker. +MEILISEARCH_URL=http://meilisearch:7700 +# Optional master/API key for secured Meilisearch instances +# MEILISEARCH_API_KEY=your_master_key_here +MEILISEARCH_INDEX_NAME=documents +ENABLE_SEARCH=True diff --git a/app/api/__init__.py b/app/api/__init__.py index 20bafc6c..b483248a 100644 --- a/app/api/__init__.py +++ b/app/api/__init__.py @@ -15,6 +15,7 @@ from app.api.logs import router as logs_router from app.api.onedrive import router as onedrive_router from app.api.openai import router as openai_router from app.api.process import router as process_router +from app.api.search import router as search_router from app.api.settings import router as settings_router from app.api.url_upload import router as url_upload_router @@ -40,3 +41,4 @@ router.include_router(google_drive_router) router.include_router(logs_router) router.include_router(settings_router) router.include_router(url_upload_router) +router.include_router(search_router) diff --git a/app/api/search.py b/app/api/search.py new file mode 100644 index 00000000..2b6a5117 --- /dev/null +++ b/app/api/search.py @@ -0,0 +1,99 @@ +"""Full-text search API endpoints. + +Provides document search across OCR text, AI metadata, filenames, and tags +via Meilisearch. Designed to serve as the backend for the UI search bar on +the /files page and as a standalone API for integrations. + +Future extension point: the OCR text stored in the index is also suitable +for RAG (Retrieval Augmented Generation) chatbot workflows. +""" + +import logging +from typing import Optional + +from fastapi import APIRouter, Query, Request + +from app.auth import require_login +from app.utils.meilisearch_client import search_documents + +logger = logging.getLogger(__name__) + +router = APIRouter() + + +@router.get("/search") +@require_login +def search_api( + request: Request, + q: str = Query(..., min_length=1, max_length=512, description="Full-text search query"), + mime_type: Optional[str] = Query(None, description="Filter by MIME type (e.g. application/pdf)"), + document_type: Optional[str] = Query(None, description="Filter by document type (e.g. Invoice)"), + language: Optional[str] = Query(None, description="Filter by language code (e.g. de, en)"), + date_from: Optional[int] = Query(None, description="Filter results created after this Unix timestamp"), + date_to: Optional[int] = Query(None, description="Filter results created before this Unix timestamp"), + page: int = Query(1, ge=1, description="Page number (1-based)"), + per_page: int = Query(20, ge=1, le=100, description="Results per page"), +): + """Search documents by full text, metadata, and tags. + + Searches across: + - Document title and filename + - OCR / extracted text + - Tags, sender, recipient, document type + - Correspondent and reference number + + Results are ranked by Meilisearch relevance and include highlighted + snippets showing where the query terms matched. + + Query Parameters: + - q: Search query (required) + - mime_type: Filter by MIME type + - document_type: Filter by document type + - language: Filter by language code + - date_from: Unix timestamp lower bound + - date_to: Unix timestamp upper bound + - page: Page number (default: 1) + - per_page: Results per page (default: 20, max: 100) + + Example: + ``` + GET /api/search?q=invoice&document_type=Invoice&date_from=1704067200&page=1&per_page=20 + ``` + + Response: + ```json + { + "results": [ + { + "file_id": 42, + "original_filename": "2026-01-15_Invoice_Amazon.pdf", + "document_title": "Amazon Invoice January 2026", + "document_type": "Invoice", + "tags": ["amazon", "invoice"], + "_formatted": { + "document_title": "Amazon Invoice January 2026", + "ocr_text": "...total amount of the invoice is..." + } + } + ], + "total": 42, + "page": 1, + "pages": 3, + "query": "invoice" + } + ``` + """ + logger.info(f"Search request: q={q!r}, mime_type={mime_type}, page={page}, per_page={per_page}") + + result = search_documents( + q, + mime_type=mime_type, + document_type=document_type, + language=language, + date_from=date_from, + date_to=date_to, + page=page, + per_page=per_page, + ) + + return result diff --git a/app/config.py b/app/config.py index 3bbae76a..41398e91 100644 --- a/app/config.py +++ b/app/config.py @@ -207,6 +207,15 @@ class Settings(BaseSettings): uptime_kuma_url: Optional[str] = None uptime_kuma_ping_interval: int = 5 # Default ping interval in minutes + # Meilisearch settings (full-text search engine) + # Default uses the Docker Compose / K8s service name so container-to-container + # networking works without any extra configuration. Override to + # "http://localhost:7700" only when running the API process outside of Docker. + meilisearch_url: str = "http://meilisearch:7700" + meilisearch_api_key: Optional[str] = None # Master or API key (optional for local dev) + meilisearch_index_name: str = "documents" + enable_search: bool = True # Enable Meilisearch full-text search integration + # HTTP request settings http_request_timeout: int = 120 # Default timeout for HTTP requests in seconds (handles large file operations) diff --git a/app/models.py b/app/models.py index 78c09df9..8997bc3c 100644 --- a/app/models.py +++ b/app/models.py @@ -55,6 +55,15 @@ class FileRecord(Base): # If this is a duplicate, record the ID of the original file for reference duplicate_of_id = Column(Integer, ForeignKey(_FILES_ID_FK), nullable=True) + # Full OCR/extracted text for full-text search and RAG + ocr_text = Column(Text, nullable=True) + + # AI-extracted metadata stored as JSON string (filename, tags, title, sender, etc.) + ai_metadata = Column(Text, nullable=True) + + # Human-readable document title from AI metadata + document_title = Column(String, nullable=True) + # Timestamp when we inserted this record created_at = Column(DateTime(timezone=True), server_default=func.now()) diff --git a/app/tasks/embed_metadata_into_pdf.py b/app/tasks/embed_metadata_into_pdf.py index 82bdb8bb..73a4d0cf 100644 --- a/app/tasks/embed_metadata_into_pdf.py +++ b/app/tasks/embed_metadata_into_pdf.py @@ -195,8 +195,26 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met original_file_path = file_record.original_file_path # Update the processed_file_path in the database file_record.processed_file_path = final_file_path + # Persist extracted text and AI metadata to DB for full-text search / RAG + file_record.ocr_text = extracted_text or None + if metadata: + try: + file_record.ai_metadata = json.dumps(metadata, ensure_ascii=False) + except Exception: + pass + file_record.document_title = ( + metadata.get("title") or metadata.get("filename") or file_record.original_filename + ) db.commit() - logger.info(f"[{task_id}] Updated database with processed_file_path: {final_file_path}") + logger.info(f"[{task_id}] Updated database with processed_file_path and search fields") + + # Index into Meilisearch for full-text search (non-blocking, best-effort) + try: + from app.utils.meilisearch_client import index_document + + index_document(file_record, extracted_text or "", metadata or {}) + except Exception as search_exc: + logger.warning(f"[{task_id}] Meilisearch indexing failed (non-fatal): {search_exc}") # Persist the metadata into a JSON file with the same base name. # Include file path references for traceability diff --git a/app/utils/meilisearch_client.py b/app/utils/meilisearch_client.py new file mode 100644 index 00000000..ff534785 --- /dev/null +++ b/app/utils/meilisearch_client.py @@ -0,0 +1,290 @@ +"""Meilisearch client utilities for full-text document search. + +This module provides functions for indexing documents into Meilisearch +and searching across OCR text, AI metadata, filenames, and tags. + +The search index is designed to support future RAG (Retrieval Augmented +Generation) workflows by storing full document text alongside structured +metadata fields. +""" + +import json +import logging +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +# Index settings applied once at index creation / first use +_INDEX_SETTINGS = { + "searchableAttributes": [ + "document_title", + "original_filename", + "ocr_text", + "tags", + "sender", + "recipient", + "document_type", + "correspondent", + ], + "filterableAttributes": [ + "mime_type", + "document_type", + "language", + "tags", + "created_at_ts", + "file_id", + ], + "sortableAttributes": [ + "created_at_ts", + "file_size", + ], + "displayedAttributes": [ + "file_id", + "original_filename", + "document_title", + "document_type", + "tags", + "sender", + "recipient", + "correspondent", + "language", + "reference_number", + "mime_type", + "file_size", + "created_at_ts", + "ocr_text", + ], + "rankingRules": [ + "words", + "typo", + "proximity", + "attribute", + "sort", + "exactness", + ], +} + + +def get_meilisearch_client(): + """Return a configured Meilisearch client, or None if unavailable/disabled.""" + try: + import meilisearch + + from app.config import settings + + if not settings.enable_search: + return None + + kwargs: dict[str, Any] = {"url": settings.meilisearch_url} + if settings.meilisearch_api_key: + kwargs["api_key"] = settings.meilisearch_api_key + + client = meilisearch.Client(**kwargs) + return client + except ImportError: + logger.warning("meilisearch package not installed; search disabled") + return None + except Exception as exc: + logger.warning(f"Could not connect to Meilisearch: {exc}") + return None + + +def _get_or_create_index(client): + """Get the documents index, creating it with settings if it doesn't exist.""" + from app.config import settings + + index_name = settings.meilisearch_index_name + try: + index = client.get_index(index_name) + except Exception: + # Index doesn't exist – create it with file_id as primary key + task = client.create_index(index_name, {"primaryKey": "file_id"}) + client.wait_for_task(task.task_uid) + index = client.get_index(index_name) + # Apply search settings + try: + task = index.update_settings(_INDEX_SETTINGS) + client.wait_for_task(task.task_uid) + except Exception as exc: + logger.warning(f"Could not update Meilisearch index settings: {exc}") + return index + + +def _build_document(file_record, text: str, metadata: dict) -> dict: + """Build a Meilisearch document from a FileRecord and extracted content.""" + import json as _json + + tags = metadata.get("tags", []) + if isinstance(tags, str): + tags = [t.strip() for t in tags.split(",") if t.strip()] + + # Unix timestamp for sorting/filtering + created_at_ts = 0 + if file_record.created_at: + try: + created_at_ts = int(file_record.created_at.timestamp()) + except Exception: + pass + + return { + "file_id": file_record.id, + "original_filename": file_record.original_filename or "", + "document_title": metadata.get("title") or metadata.get("filename") or file_record.original_filename or "", + "document_type": metadata.get("document_type") or metadata.get("kommunikationsart") or "", + "tags": tags, + "sender": metadata.get("absender") or "", + "recipient": metadata.get("empfaenger") or "", + "correspondent": metadata.get("correspondent") or "", + "language": metadata.get("language") or "", + "reference_number": metadata.get("reference_number") or "", + "mime_type": file_record.mime_type or "", + "file_size": file_record.file_size or 0, + "created_at_ts": created_at_ts, + "ocr_text": text or "", + } + + +def index_document(file_record, text: str, metadata: dict) -> bool: + """Index a document in Meilisearch. + + Args: + file_record: FileRecord ORM instance with at minimum .id set. + text: Full OCR / extracted text for the document. + metadata: AI-extracted metadata dict. + + Returns: + True if indexing succeeded, False otherwise. + """ + client = get_meilisearch_client() + if client is None: + return False + + try: + index = _get_or_create_index(client) + doc = _build_document(file_record, text, metadata) + task = index.add_documents([doc]) + logger.info(f"Queued Meilisearch indexing for file_id={file_record.id} (task_uid={task.task_uid})") + return True + except Exception as exc: + logger.warning(f"Meilisearch indexing failed for file_id={file_record.id}: {exc}") + return False + + +def delete_document(file_id: int) -> bool: + """Remove a document from the Meilisearch index. + + Args: + file_id: The database ID of the file to remove. + + Returns: + True if deletion succeeded, False otherwise. + """ + client = get_meilisearch_client() + if client is None: + return False + + try: + from app.config import settings + + index = client.get_index(settings.meilisearch_index_name) + task = index.delete_document(file_id) + logger.info(f"Queued Meilisearch deletion for file_id={file_id} (task_uid={task.task_uid})") + return True + except Exception as exc: + logger.warning(f"Meilisearch deletion failed for file_id={file_id}: {exc}") + return False + + +def search_documents( + query: str, + *, + mime_type: Optional[str] = None, + document_type: Optional[str] = None, + language: Optional[str] = None, + date_from: Optional[int] = None, + date_to: Optional[int] = None, + page: int = 1, + per_page: int = 20, +) -> dict: + """Search documents in Meilisearch. + + Args: + query: Full-text search query string. + mime_type: Optional MIME-type filter. + document_type: Optional document type filter. + language: Optional language filter (ISO 639-1, e.g. "de"). + date_from: Optional lower bound Unix timestamp for created_at. + date_to: Optional upper bound Unix timestamp for created_at. + page: 1-based page number. + per_page: Results per page (max 100). + + Returns: + Dict with keys: results, total, page, pages, query. + Returns empty results dict on any error. + """ + empty: dict = {"results": [], "total": 0, "page": page, "pages": 0, "query": query} + + client = get_meilisearch_client() + if client is None: + return empty + + try: + from app.config import settings + + index = _get_or_create_index(client) + + # Build filter expressions + filters: list[str] = [] + if mime_type: + filters.append(f'mime_type = "{mime_type}"') + if document_type: + filters.append(f'document_type = "{document_type}"') + if language: + filters.append(f'language = "{language}"') + if date_from is not None: + filters.append(f"created_at_ts >= {date_from}") + if date_to is not None: + filters.append(f"created_at_ts <= {date_to}") + + search_params: dict[str, Any] = { + "offset": (page - 1) * per_page, + "limit": per_page, + "attributesToHighlight": ["document_title", "original_filename", "ocr_text", "tags"], + "highlightPreTag": "", + "highlightPostTag": "", + "attributesToCrop": ["ocr_text"], + "cropLength": 200, + } + + if filters: + search_params["filter"] = " AND ".join(filters) + + result = index.search(query, search_params) + + hits = result.get("hits", []) + total = result.get("estimatedTotalHits", result.get("nbHits", len(hits))) + + # Attach highlights to each hit + formatted_results = [] + for hit in hits: + formatted = dict(hit) + # Include formatted (highlighted) snippets if available + if "_formatted" in hit: + formatted["_formatted"] = hit["_formatted"] + # Exclude raw ocr_text from results (use _formatted snippet instead) + formatted.pop("ocr_text", None) + formatted_results.append(formatted) + + pages = (total + per_page - 1) // per_page if total > 0 else 0 + + return { + "results": formatted_results, + "total": total, + "page": page, + "pages": pages, + "query": query, + } + + except Exception as exc: + logger.warning(f"Meilisearch search failed for query '{query}': {exc}") + return empty diff --git a/docker-compose.yaml b/docker-compose.yaml index e09c5b23..06e0f1c8 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -59,6 +59,15 @@ services: container_name: gotenberg restart: always + meilisearch: + image: getmeili/meilisearch:latest + container_name: document_meilisearch + restart: always + environment: + - MEILI_NO_ANALYTICS=true + volumes: + - /var/docparse/meilisearch:/meili_data + redis: image: redis:alpine container_name: document_redis diff --git a/frontend/templates/files.html b/frontend/templates/files.html index d277d30e..ab44dbb5 100644 --- a/frontend/templates/files.html +++ b/frontend/templates/files.html @@ -418,6 +418,47 @@ + +
+
+ +
+ + + +
+
+
+ + + +