fix(config): default meilisearch_url to http://meilisearch:7700 for Docker/K8s service discovery

Co-authored-by: christianlouis <361235+christianlouis@users.noreply.github.com>
This commit is contained in:
copilot-swe-agent[bot]
2026-02-25 13:27:09 +00:00
parent cc4d26aacd
commit ff96340154
12 changed files with 946 additions and 2 deletions
+11
View File
@@ -299,3 +299,14 @@ NOTIFY_ON_FILE_PROCESSED=True
# Uptime Kuma
UPTIME_KUMA_URL=https://status.example.com/api/push/abcdef123456?status=up
UPTIME_KUMA_PING_INTERVAL=5
# **Full-Text Search (Meilisearch)**
# URL for the Meilisearch instance.
# Default is "http://meilisearch:7700" — the Docker Compose / K8s service name —
# so container-to-container networking works without extra configuration.
# Override to "http://localhost:7700" only when running the API process outside Docker.
MEILISEARCH_URL=http://meilisearch:7700
# Optional master/API key for secured Meilisearch instances
# MEILISEARCH_API_KEY=your_master_key_here
MEILISEARCH_INDEX_NAME=documents
ENABLE_SEARCH=True
+2
View File
@@ -15,6 +15,7 @@ from app.api.logs import router as logs_router
from app.api.onedrive import router as onedrive_router
from app.api.openai import router as openai_router
from app.api.process import router as process_router
from app.api.search import router as search_router
from app.api.settings import router as settings_router
from app.api.url_upload import router as url_upload_router
@@ -40,3 +41,4 @@ router.include_router(google_drive_router)
router.include_router(logs_router)
router.include_router(settings_router)
router.include_router(url_upload_router)
router.include_router(search_router)
+99
View File
@@ -0,0 +1,99 @@
"""Full-text search API endpoints.
Provides document search across OCR text, AI metadata, filenames, and tags
via Meilisearch. Designed to serve as the backend for the UI search bar on
the /files page and as a standalone API for integrations.
Future extension point: the OCR text stored in the index is also suitable
for RAG (Retrieval Augmented Generation) chatbot workflows.
"""
import logging
from typing import Optional
from fastapi import APIRouter, Query, Request
from app.auth import require_login
from app.utils.meilisearch_client import search_documents
logger = logging.getLogger(__name__)
router = APIRouter()
@router.get("/search")
@require_login
def search_api(
request: Request,
q: str = Query(..., min_length=1, max_length=512, description="Full-text search query"),
mime_type: Optional[str] = Query(None, description="Filter by MIME type (e.g. application/pdf)"),
document_type: Optional[str] = Query(None, description="Filter by document type (e.g. Invoice)"),
language: Optional[str] = Query(None, description="Filter by language code (e.g. de, en)"),
date_from: Optional[int] = Query(None, description="Filter results created after this Unix timestamp"),
date_to: Optional[int] = Query(None, description="Filter results created before this Unix timestamp"),
page: int = Query(1, ge=1, description="Page number (1-based)"),
per_page: int = Query(20, ge=1, le=100, description="Results per page"),
):
"""Search documents by full text, metadata, and tags.
Searches across:
- Document title and filename
- OCR / extracted text
- Tags, sender, recipient, document type
- Correspondent and reference number
Results are ranked by Meilisearch relevance and include highlighted
snippets showing where the query terms matched.
Query Parameters:
- q: Search query (required)
- mime_type: Filter by MIME type
- document_type: Filter by document type
- language: Filter by language code
- date_from: Unix timestamp lower bound
- date_to: Unix timestamp upper bound
- page: Page number (default: 1)
- per_page: Results per page (default: 20, max: 100)
Example:
```
GET /api/search?q=invoice&document_type=Invoice&date_from=1704067200&page=1&per_page=20
```
Response:
```json
{
"results": [
{
"file_id": 42,
"original_filename": "2026-01-15_Invoice_Amazon.pdf",
"document_title": "Amazon Invoice January 2026",
"document_type": "Invoice",
"tags": ["amazon", "invoice"],
"_formatted": {
"document_title": "Amazon <mark>Invoice</mark> January 2026",
"ocr_text": "...total amount of the <mark>invoice</mark> is..."
}
}
],
"total": 42,
"page": 1,
"pages": 3,
"query": "invoice"
}
```
"""
logger.info(f"Search request: q={q!r}, mime_type={mime_type}, page={page}, per_page={per_page}")
result = search_documents(
q,
mime_type=mime_type,
document_type=document_type,
language=language,
date_from=date_from,
date_to=date_to,
page=page,
per_page=per_page,
)
return result
+9
View File
@@ -207,6 +207,15 @@ class Settings(BaseSettings):
uptime_kuma_url: Optional[str] = None
uptime_kuma_ping_interval: int = 5 # Default ping interval in minutes
# Meilisearch settings (full-text search engine)
# Default uses the Docker Compose / K8s service name so container-to-container
# networking works without any extra configuration. Override to
# "http://localhost:7700" only when running the API process outside of Docker.
meilisearch_url: str = "http://meilisearch:7700"
meilisearch_api_key: Optional[str] = None # Master or API key (optional for local dev)
meilisearch_index_name: str = "documents"
enable_search: bool = True # Enable Meilisearch full-text search integration
# HTTP request settings
http_request_timeout: int = 120 # Default timeout for HTTP requests in seconds (handles large file operations)
+9
View File
@@ -55,6 +55,15 @@ class FileRecord(Base):
# If this is a duplicate, record the ID of the original file for reference
duplicate_of_id = Column(Integer, ForeignKey(_FILES_ID_FK), nullable=True)
# Full OCR/extracted text for full-text search and RAG
ocr_text = Column(Text, nullable=True)
# AI-extracted metadata stored as JSON string (filename, tags, title, sender, etc.)
ai_metadata = Column(Text, nullable=True)
# Human-readable document title from AI metadata
document_title = Column(String, nullable=True)
# Timestamp when we inserted this record
created_at = Column(DateTime(timezone=True), server_default=func.now())
+19 -1
View File
@@ -195,8 +195,26 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met
original_file_path = file_record.original_file_path
# Update the processed_file_path in the database
file_record.processed_file_path = final_file_path
# Persist extracted text and AI metadata to DB for full-text search / RAG
file_record.ocr_text = extracted_text or None
if metadata:
try:
file_record.ai_metadata = json.dumps(metadata, ensure_ascii=False)
except Exception:
pass
file_record.document_title = (
metadata.get("title") or metadata.get("filename") or file_record.original_filename
)
db.commit()
logger.info(f"[{task_id}] Updated database with processed_file_path: {final_file_path}")
logger.info(f"[{task_id}] Updated database with processed_file_path and search fields")
# Index into Meilisearch for full-text search (non-blocking, best-effort)
try:
from app.utils.meilisearch_client import index_document
index_document(file_record, extracted_text or "", metadata or {})
except Exception as search_exc:
logger.warning(f"[{task_id}] Meilisearch indexing failed (non-fatal): {search_exc}")
# Persist the metadata into a JSON file with the same base name.
# Include file path references for traceability
+290
View File
@@ -0,0 +1,290 @@
"""Meilisearch client utilities for full-text document search.
This module provides functions for indexing documents into Meilisearch
and searching across OCR text, AI metadata, filenames, and tags.
The search index is designed to support future RAG (Retrieval Augmented
Generation) workflows by storing full document text alongside structured
metadata fields.
"""
import json
import logging
from typing import Any, Optional
logger = logging.getLogger(__name__)
# Index settings applied once at index creation / first use
_INDEX_SETTINGS = {
"searchableAttributes": [
"document_title",
"original_filename",
"ocr_text",
"tags",
"sender",
"recipient",
"document_type",
"correspondent",
],
"filterableAttributes": [
"mime_type",
"document_type",
"language",
"tags",
"created_at_ts",
"file_id",
],
"sortableAttributes": [
"created_at_ts",
"file_size",
],
"displayedAttributes": [
"file_id",
"original_filename",
"document_title",
"document_type",
"tags",
"sender",
"recipient",
"correspondent",
"language",
"reference_number",
"mime_type",
"file_size",
"created_at_ts",
"ocr_text",
],
"rankingRules": [
"words",
"typo",
"proximity",
"attribute",
"sort",
"exactness",
],
}
def get_meilisearch_client():
"""Return a configured Meilisearch client, or None if unavailable/disabled."""
try:
import meilisearch
from app.config import settings
if not settings.enable_search:
return None
kwargs: dict[str, Any] = {"url": settings.meilisearch_url}
if settings.meilisearch_api_key:
kwargs["api_key"] = settings.meilisearch_api_key
client = meilisearch.Client(**kwargs)
return client
except ImportError:
logger.warning("meilisearch package not installed; search disabled")
return None
except Exception as exc:
logger.warning(f"Could not connect to Meilisearch: {exc}")
return None
def _get_or_create_index(client):
"""Get the documents index, creating it with settings if it doesn't exist."""
from app.config import settings
index_name = settings.meilisearch_index_name
try:
index = client.get_index(index_name)
except Exception:
# Index doesn't exist create it with file_id as primary key
task = client.create_index(index_name, {"primaryKey": "file_id"})
client.wait_for_task(task.task_uid)
index = client.get_index(index_name)
# Apply search settings
try:
task = index.update_settings(_INDEX_SETTINGS)
client.wait_for_task(task.task_uid)
except Exception as exc:
logger.warning(f"Could not update Meilisearch index settings: {exc}")
return index
def _build_document(file_record, text: str, metadata: dict) -> dict:
"""Build a Meilisearch document from a FileRecord and extracted content."""
import json as _json
tags = metadata.get("tags", [])
if isinstance(tags, str):
tags = [t.strip() for t in tags.split(",") if t.strip()]
# Unix timestamp for sorting/filtering
created_at_ts = 0
if file_record.created_at:
try:
created_at_ts = int(file_record.created_at.timestamp())
except Exception:
pass
return {
"file_id": file_record.id,
"original_filename": file_record.original_filename or "",
"document_title": metadata.get("title") or metadata.get("filename") or file_record.original_filename or "",
"document_type": metadata.get("document_type") or metadata.get("kommunikationsart") or "",
"tags": tags,
"sender": metadata.get("absender") or "",
"recipient": metadata.get("empfaenger") or "",
"correspondent": metadata.get("correspondent") or "",
"language": metadata.get("language") or "",
"reference_number": metadata.get("reference_number") or "",
"mime_type": file_record.mime_type or "",
"file_size": file_record.file_size or 0,
"created_at_ts": created_at_ts,
"ocr_text": text or "",
}
def index_document(file_record, text: str, metadata: dict) -> bool:
"""Index a document in Meilisearch.
Args:
file_record: FileRecord ORM instance with at minimum .id set.
text: Full OCR / extracted text for the document.
metadata: AI-extracted metadata dict.
Returns:
True if indexing succeeded, False otherwise.
"""
client = get_meilisearch_client()
if client is None:
return False
try:
index = _get_or_create_index(client)
doc = _build_document(file_record, text, metadata)
task = index.add_documents([doc])
logger.info(f"Queued Meilisearch indexing for file_id={file_record.id} (task_uid={task.task_uid})")
return True
except Exception as exc:
logger.warning(f"Meilisearch indexing failed for file_id={file_record.id}: {exc}")
return False
def delete_document(file_id: int) -> bool:
"""Remove a document from the Meilisearch index.
Args:
file_id: The database ID of the file to remove.
Returns:
True if deletion succeeded, False otherwise.
"""
client = get_meilisearch_client()
if client is None:
return False
try:
from app.config import settings
index = client.get_index(settings.meilisearch_index_name)
task = index.delete_document(file_id)
logger.info(f"Queued Meilisearch deletion for file_id={file_id} (task_uid={task.task_uid})")
return True
except Exception as exc:
logger.warning(f"Meilisearch deletion failed for file_id={file_id}: {exc}")
return False
def search_documents(
query: str,
*,
mime_type: Optional[str] = None,
document_type: Optional[str] = None,
language: Optional[str] = None,
date_from: Optional[int] = None,
date_to: Optional[int] = None,
page: int = 1,
per_page: int = 20,
) -> dict:
"""Search documents in Meilisearch.
Args:
query: Full-text search query string.
mime_type: Optional MIME-type filter.
document_type: Optional document type filter.
language: Optional language filter (ISO 639-1, e.g. "de").
date_from: Optional lower bound Unix timestamp for created_at.
date_to: Optional upper bound Unix timestamp for created_at.
page: 1-based page number.
per_page: Results per page (max 100).
Returns:
Dict with keys: results, total, page, pages, query.
Returns empty results dict on any error.
"""
empty: dict = {"results": [], "total": 0, "page": page, "pages": 0, "query": query}
client = get_meilisearch_client()
if client is None:
return empty
try:
from app.config import settings
index = _get_or_create_index(client)
# Build filter expressions
filters: list[str] = []
if mime_type:
filters.append(f'mime_type = "{mime_type}"')
if document_type:
filters.append(f'document_type = "{document_type}"')
if language:
filters.append(f'language = "{language}"')
if date_from is not None:
filters.append(f"created_at_ts >= {date_from}")
if date_to is not None:
filters.append(f"created_at_ts <= {date_to}")
search_params: dict[str, Any] = {
"offset": (page - 1) * per_page,
"limit": per_page,
"attributesToHighlight": ["document_title", "original_filename", "ocr_text", "tags"],
"highlightPreTag": "<mark>",
"highlightPostTag": "</mark>",
"attributesToCrop": ["ocr_text"],
"cropLength": 200,
}
if filters:
search_params["filter"] = " AND ".join(filters)
result = index.search(query, search_params)
hits = result.get("hits", [])
total = result.get("estimatedTotalHits", result.get("nbHits", len(hits)))
# Attach highlights to each hit
formatted_results = []
for hit in hits:
formatted = dict(hit)
# Include formatted (highlighted) snippets if available
if "_formatted" in hit:
formatted["_formatted"] = hit["_formatted"]
# Exclude raw ocr_text from results (use _formatted snippet instead)
formatted.pop("ocr_text", None)
formatted_results.append(formatted)
pages = (total + per_page - 1) // per_page if total > 0 else 0
return {
"results": formatted_results,
"total": total,
"page": page,
"pages": pages,
"query": query,
}
except Exception as exc:
logger.warning(f"Meilisearch search failed for query '{query}': {exc}")
return empty
+9
View File
@@ -59,6 +59,15 @@ services:
container_name: gotenberg
restart: always
meilisearch:
image: getmeili/meilisearch:latest
container_name: document_meilisearch
restart: always
environment:
- MEILI_NO_ANALYTICS=true
volumes:
- /var/docparse/meilisearch:/meili_data
redis:
image: redis:alpine
container_name: document_redis
+152
View File
@@ -418,6 +418,47 @@
</form>
</div>
<!-- Full-Text Search Section -->
<div class="filters-section" style="margin-top: 0.75rem;">
<div style="width: 100%;">
<label for="fulltext-search" style="font-weight: 600; display: block; margin-bottom: 0.4rem;">
<i class="fas fa-search"></i> Full-Text Search (OCR text, metadata, tags)
</label>
<div style="display: flex; gap: 0.5rem; align-items: center; flex-wrap: wrap;">
<input
type="text"
id="fulltext-search"
placeholder="Search document content, sender, tags, type..."
style="flex: 1; min-width: 220px; padding: 0.5rem 0.75rem; border: 1px solid #d1d5db; border-radius: 0.375rem; font-size: 0.875rem;"
oninput="debounceSearch(this.value)"
>
<button
type="button"
onclick="runFullTextSearch()"
style="padding: 0.5rem 1rem; background-color: #3182ce; color: white; border: none; border-radius: 0.375rem; cursor: pointer; font-size: 0.875rem; white-space: nowrap;"
>
Search
</button>
<button
type="button"
onclick="clearFullTextSearch()"
style="padding: 0.5rem 1rem; background-color: #6b7280; color: white; border: none; border-radius: 0.375rem; cursor: pointer; font-size: 0.875rem; white-space: nowrap;"
>
Clear
</button>
</div>
</div>
</div>
<!-- Full-Text Search Results Panel -->
<div id="search-results-panel" style="display: none; margin-top: 0.5rem; border: 1px solid #e5e7eb; border-radius: 0.5rem; background: white; box-shadow: 0 1px 3px rgba(0,0,0,0.08);">
<div style="padding: 0.75rem 1rem; border-bottom: 1px solid #e5e7eb; display: flex; justify-content: space-between; align-items: center; background: #f9fafb; border-radius: 0.5rem 0.5rem 0 0;">
<span id="search-results-summary" style="font-size: 0.875rem; color: #374151;"></span>
<div id="search-results-pagination" style="display: flex; gap: 0.5rem;"></div>
</div>
<div id="search-results-list" style="padding: 0.5rem 0;"></div>
</div>
<!-- Bulk Actions Section -->
<div id="bulkActionsBar" class="filters-section" style="display: none; background-color: #e6f3ff;">
<div class="flex flex-col sm:flex-row sm:justify-between sm:items-center gap-3">
@@ -887,6 +928,117 @@
function closeUploadModal() {
uploadModal.classList.remove('active');
}
// ---- Full-Text Search (Meilisearch) ----
let _searchDebounceTimer = null;
let _searchCurrentPage = 1;
function debounceSearch(value) {
clearTimeout(_searchDebounceTimer);
if (!value || value.trim().length < 2) {
clearFullTextSearch();
return;
}
_searchDebounceTimer = setTimeout(() => {
_searchCurrentPage = 1;
runFullTextSearch();
}, 400);
}
function runFullTextSearch(page) {
const input = document.getElementById('fulltext-search');
const q = input ? input.value.trim() : '';
if (!q) { clearFullTextSearch(); return; }
if (page) _searchCurrentPage = page;
const panel = document.getElementById('search-results-panel');
const list = document.getElementById('search-results-list');
const summary = document.getElementById('search-results-summary');
const pagination = document.getElementById('search-results-pagination');
list.innerHTML = '<div style="padding: 1rem; color: #6b7280; font-size: 0.875rem;"><i class="fas fa-spinner fa-spin"></i> Searching…</div>';
summary.textContent = '';
pagination.innerHTML = '';
panel.style.display = 'block';
const params = new URLSearchParams({ q, page: _searchCurrentPage, per_page: 20 });
fetch(`/api/search?${params.toString()}`)
.then(r => {
if (!r.ok) throw new Error(`Search returned ${r.status}`);
return r.json();
})
.then(data => renderSearchResults(data, q))
.catch(err => {
list.innerHTML = `<div style="padding: 1rem; color: #dc2626; font-size: 0.875rem;"><i class="fas fa-exclamation-triangle"></i> Search unavailable: ${err.message}</div>`;
summary.textContent = '';
});
}
function renderSearchResults(data, q) {
const panel = document.getElementById('search-results-panel');
const list = document.getElementById('search-results-list');
const summary = document.getElementById('search-results-summary');
const pagination = document.getElementById('search-results-pagination');
const { results, total, page, pages } = data;
summary.textContent = `${total} result${total !== 1 ? 's' : ''} for "${q}"`;
if (!results || results.length === 0) {
list.innerHTML = '<div style="padding: 1rem; color: #6b7280; font-size: 0.875rem;">No results found.</div>';
pagination.innerHTML = '';
return;
}
list.innerHTML = results.map(hit => {
const fmt = hit._formatted || {};
const title = fmt.document_title || hit.document_title || hit.original_filename || '(untitled)';
const filename = fmt.original_filename || hit.original_filename || '';
const snippet = fmt.ocr_text || '';
const tags = Array.isArray(hit.tags) ? hit.tags.join(', ') : (hit.tags || '');
const docType = hit.document_type || '';
return `<div style="padding: 0.75rem 1rem; border-bottom: 1px solid #f3f4f6; display: flex; gap: 0.75rem; align-items: flex-start;">
<div style="flex-shrink: 0; color: #3b82f6; font-size: 1.25rem; padding-top: 0.1rem;">
<i class="fas fa-file-pdf"></i>
</div>
<div style="flex: 1; min-width: 0;">
<div style="font-weight: 600; font-size: 0.9rem; color: #111827;">${title}</div>
${filename ? `<div style="font-size: 0.8rem; color: #6b7280; margin-top: 0.15rem;">${filename}</div>` : ''}
${docType ? `<span style="display: inline-block; margin-top: 0.25rem; padding: 0.1rem 0.5rem; background: #eff6ff; color: #1d4ed8; border-radius: 9999px; font-size: 0.75rem;">${docType}</span>` : ''}
${tags ? `<span style="display: inline-block; margin-top: 0.25rem; margin-left: 0.25rem; padding: 0.1rem 0.5rem; background: #f0fdf4; color: #15803d; border-radius: 9999px; font-size: 0.75rem;">${tags}</span>` : ''}
${snippet ? `<div style="margin-top: 0.4rem; font-size: 0.8rem; color: #374151; white-space: pre-wrap; word-break: break-word;">…${snippet}…</div>` : ''}
</div>
<div style="flex-shrink: 0;">
<a href="/files/${hit.file_id}" style="padding: 0.25rem 0.6rem; background: #f3f4f6; color: #374151; border-radius: 0.25rem; font-size: 0.8rem; text-decoration: none; white-space: nowrap;" title="View file">
<i class="fas fa-external-link-alt"></i>
</a>
</div>
</div>`;
}).join('');
// Pagination
if (pages > 1) {
const btns = [];
if (page > 1) {
btns.push(`<button onclick="runFullTextSearch(${page - 1})" style="padding: 0.25rem 0.6rem; border: 1px solid #d1d5db; border-radius: 0.25rem; font-size: 0.8rem; cursor: pointer; background: white;">« Prev</button>`);
}
btns.push(`<span style="padding: 0.25rem 0.6rem; font-size: 0.8rem; color: #6b7280;">Page ${page} / ${pages}</span>`);
if (page < pages) {
btns.push(`<button onclick="runFullTextSearch(${page + 1})" style="padding: 0.25rem 0.6rem; border: 1px solid #d1d5db; border-radius: 0.25rem; font-size: 0.8rem; cursor: pointer; background: white;">Next »</button>`);
}
pagination.innerHTML = btns.join('');
} else {
pagination.innerHTML = '';
}
}
function clearFullTextSearch() {
const panel = document.getElementById('search-results-panel');
const input = document.getElementById('fulltext-search');
if (panel) panel.style.display = 'none';
if (input) input.value = '';
_searchCurrentPage = 1;
}
</script>
</div>
{% endblock %}
@@ -0,0 +1,31 @@
"""Add OCR text and AI metadata fields to files table for full-text search
Revision ID: 004_add_search_fields
Revises: 003_add_deduplication_support
Create Date: 2026-02-25
"""
from typing import Union
import sqlalchemy as sa
from alembic import op
# revision identifiers, used by Alembic.
revision: str = "004_add_search_fields"
down_revision: Union[str, None] = "003_add_deduplication_support"
depends_on: Union[str, None] = None
def upgrade() -> None:
"""Add ocr_text, ai_metadata, and document_title columns to files table."""
op.add_column("files", sa.Column("ocr_text", sa.Text(), nullable=True))
op.add_column("files", sa.Column("ai_metadata", sa.Text(), nullable=True))
op.add_column("files", sa.Column("document_title", sa.String(), nullable=True))
def downgrade() -> None:
"""Remove ocr_text, ai_metadata, and document_title columns from files table."""
op.drop_column("files", "document_title")
op.drop_column("files", "ai_metadata")
op.drop_column("files", "ocr_text")
+1
View File
@@ -43,3 +43,4 @@ litellm>=1.0.0,<2.0.0
pytesseract>=0.3.10 # Python wrapper for Tesseract OCR
pdf2image>=1.17.0 # Convert PDF pages to images (used by Tesseract and EasyOCR providers)
ocrmypdf>=16.0.0,<17.0.0 # Post-processing: embeds searchable text layers into PDFs via Tesseract
meilisearch>=0.31.0 # Full-text search engine client
+313
View File
@@ -0,0 +1,313 @@
"""Tests for the full-text search API (app/api/search.py) and Meilisearch
client utilities (app/utils/meilisearch_client.py).
"""
from unittest.mock import MagicMock, patch
import pytest
# ---------------------------------------------------------------------------
# app/utils/meilisearch_client tests
# ---------------------------------------------------------------------------
@pytest.mark.unit
class TestMeilisearchClientDisabled:
"""Tests when search is disabled or Meilisearch is unavailable."""
def test_get_client_returns_none_when_disabled(self):
"""get_meilisearch_client returns None when enable_search=False."""
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=None):
from app.utils.meilisearch_client import index_document, search_documents
class _FakeRecord:
id = 1
original_filename = "test.pdf"
mime_type = "application/pdf"
file_size = 1024
created_at = None
result = index_document(_FakeRecord(), "some text", {})
assert result is False
result = search_documents("invoice")
assert result["results"] == []
assert result["total"] == 0
def test_search_documents_import_error(self):
"""search_documents returns empty dict when meilisearch not installed."""
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=None):
from app.utils.meilisearch_client import search_documents
result = search_documents("test")
assert result["results"] == []
assert result["total"] == 0
assert result["page"] == 1
assert result["query"] == "test"
def test_delete_document_no_client(self):
"""delete_document returns False when client unavailable."""
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=None):
from app.utils.meilisearch_client import delete_document
result = delete_document(99)
assert result is False
@pytest.mark.unit
class TestMeilisearchIndexDocument:
"""Tests for index_document function."""
def test_index_document_success(self):
"""index_document returns True when Meilisearch succeeds."""
mock_client = MagicMock()
mock_index = MagicMock()
mock_task = MagicMock()
mock_task.task_uid = 1
mock_client.get_index.return_value = mock_index
mock_index.add_documents.return_value = mock_task
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import index_document
class _FakeRecord:
id = 1
original_filename = "invoice.pdf"
mime_type = "application/pdf"
file_size = 2048
created_at = None
result = index_document(
_FakeRecord(),
"This is an invoice for services rendered",
{
"title": "Invoice January 2026",
"document_type": "Invoice",
"tags": ["invoice", "services"],
"absender": "ACME Corp",
"language": "en",
},
)
assert result is True
mock_index.add_documents.assert_called_once()
call_docs = mock_index.add_documents.call_args[0][0]
assert len(call_docs) == 1
doc = call_docs[0]
assert doc["file_id"] == 1
assert doc["document_title"] == "Invoice January 2026"
assert "invoice" in doc["tags"]
def test_index_document_meilisearch_error(self):
"""index_document returns False on Meilisearch exception."""
mock_client = MagicMock()
mock_index = MagicMock()
mock_client.get_index.return_value = mock_index
mock_index.add_documents.side_effect = RuntimeError("Meilisearch down")
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import index_document
class _FakeRecord:
id = 2
original_filename = "test.pdf"
mime_type = "application/pdf"
file_size = 512
created_at = None
result = index_document(_FakeRecord(), "some text", {})
assert result is False
@pytest.mark.unit
class TestMeilisearchSearchDocuments:
"""Tests for search_documents function."""
def _make_mock_client(self, hits=None, total=None):
mock_client = MagicMock()
mock_index = MagicMock()
mock_client.get_index.return_value = mock_index
mock_index.search.return_value = {
"hits": hits or [],
"estimatedTotalHits": total if total is not None else len(hits or []),
}
return mock_client, mock_index
def test_search_returns_results(self):
"""search_documents returns hits from Meilisearch."""
hits = [
{
"file_id": 42,
"original_filename": "2026-01-15_Invoice_Amazon.pdf",
"document_title": "Amazon Invoice",
"document_type": "Invoice",
"tags": ["amazon", "invoice"],
"ocr_text": "Amazon invoice content here",
"_formatted": {
"document_title": "Amazon <mark>Invoice</mark>",
"ocr_text": "…Amazon <mark>invoice</mark> content here…",
},
}
]
mock_client, mock_index = self._make_mock_client(hits=hits, total=1)
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import search_documents
result = search_documents("invoice", page=1, per_page=20)
assert result["total"] == 1
assert result["pages"] == 1
assert result["query"] == "invoice"
assert len(result["results"]) == 1
# Raw ocr_text should be stripped from result (only _formatted snippet kept)
assert "ocr_text" not in result["results"][0]
assert result["results"][0]["file_id"] == 42
def test_search_with_filters(self):
"""search_documents passes filter expressions to Meilisearch."""
mock_client, mock_index = self._make_mock_client(hits=[], total=0)
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import search_documents
search_documents("contract", mime_type="application/pdf", language="de", page=1, per_page=10)
call_kwargs = mock_index.search.call_args
search_params = call_kwargs[0][1]
assert "filter" in search_params
assert 'mime_type = "application/pdf"' in search_params["filter"]
assert 'language = "de"' in search_params["filter"]
def test_search_pagination(self):
"""search_documents applies correct offset for page 2."""
mock_client, mock_index = self._make_mock_client(hits=[], total=50)
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import search_documents
result = search_documents("test", page=3, per_page=10)
call_kwargs = mock_index.search.call_args
search_params = call_kwargs[0][1]
assert search_params["offset"] == 20 # (3-1) * 10
assert search_params["limit"] == 10
assert result["pages"] == 5
def test_search_empty_results(self):
"""search_documents returns proper empty structure."""
mock_client, mock_index = self._make_mock_client(hits=[], total=0)
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import search_documents
result = search_documents("nothing")
assert result["results"] == []
assert result["total"] == 0
assert result["pages"] == 0
def test_search_exception_returns_empty(self):
"""search_documents returns empty dict on Meilisearch exception."""
mock_client = MagicMock()
mock_index = MagicMock()
mock_client.get_index.side_effect = RuntimeError("Meilisearch unavailable")
with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client):
from app.utils.meilisearch_client import search_documents
result = search_documents("invoice")
assert result["results"] == []
assert result["total"] == 0
# ---------------------------------------------------------------------------
# Search API endpoint tests (GET /api/search)
# ---------------------------------------------------------------------------
@pytest.mark.unit
class TestSearchAPIEndpoint:
"""Tests for GET /api/search endpoint."""
def test_search_endpoint_success(self, client):
"""GET /api/search?q=... returns search results."""
mock_result = {
"results": [{"file_id": 1, "document_title": "Test Invoice", "document_type": "Invoice"}],
"total": 1,
"page": 1,
"pages": 1,
"query": "invoice",
}
with patch("app.api.search.search_documents", return_value=mock_result):
response = client.get("/api/search?q=invoice")
assert response.status_code == 200
data = response.json()
assert data["total"] == 1
assert data["query"] == "invoice"
assert len(data["results"]) == 1
def test_search_endpoint_missing_query(self, client):
"""GET /api/search without q returns 422."""
response = client.get("/api/search")
assert response.status_code == 422
def test_search_endpoint_empty_query(self, client):
"""GET /api/search?q= (empty) returns 422 due to min_length=1."""
response = client.get("/api/search?q=")
assert response.status_code == 422
def test_search_endpoint_with_filters(self, client):
"""GET /api/search with optional filters passes them to search_documents."""
mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "invoice"}
with patch("app.api.search.search_documents", return_value=mock_result) as mock_search:
response = client.get("/api/search?q=invoice&mime_type=application/pdf&language=en&page=2&per_page=10")
assert response.status_code == 200
mock_search.assert_called_once_with(
"invoice",
mime_type="application/pdf",
document_type=None,
language="en",
date_from=None,
date_to=None,
page=2,
per_page=10,
)
def test_search_endpoint_per_page_max(self, client):
"""GET /api/search with per_page > 100 returns 422."""
response = client.get("/api/search?q=test&per_page=200")
assert response.status_code == 422
def test_search_endpoint_pagination_defaults(self, client):
"""GET /api/search uses default page=1 per_page=20."""
mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "test"}
with patch("app.api.search.search_documents", return_value=mock_result) as mock_search:
response = client.get("/api/search?q=test")
assert response.status_code == 200
mock_search.assert_called_once_with(
"test",
mime_type=None,
document_type=None,
language=None,
date_from=None,
date_to=None,
page=1,
per_page=20,
)
def test_search_endpoint_date_filters(self, client):
"""GET /api/search with date_from and date_to passes them as int."""
mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "contract"}
with patch("app.api.search.search_documents", return_value=mock_result) as mock_search:
response = client.get("/api/search?q=contract&date_from=1704067200&date_to=1735689600")
assert response.status_code == 200
call_kwargs = mock_search.call_args
assert call_kwargs[1]["date_from"] == 1704067200
assert call_kwargs[1]["date_to"] == 1735689600