diff --git a/app/api/saved_searches.py b/app/api/saved_searches.py index 92caba31..20c1f004 100644 --- a/app/api/saved_searches.py +++ b/app/api/saved_searches.py @@ -23,10 +23,14 @@ router = APIRouter(prefix="/saved-searches", tags=["saved-searches"]) DbSession = Annotated[Session, Depends(get_db)] -# Allowed filter keys that can be saved +# Allowed filter keys that can be saved. +# Files-view keys: search, mime_type, status, storage_provider, sort_by, sort_order +# Search-view keys: q, document_type, language, sender, text_quality +# Shared keys: tags, date_from, date_to ALLOWED_FILTER_KEYS = frozenset( { "search", + "q", "mime_type", "status", "date_from", @@ -35,6 +39,10 @@ ALLOWED_FILTER_KEYS = frozenset( "tags", "sort_by", "sort_order", + "document_type", + "language", + "sender", + "text_quality", } ) diff --git a/app/api/search.py b/app/api/search.py index 2b6a5117..0cec03de 100644 --- a/app/api/search.py +++ b/app/api/search.py @@ -29,6 +29,12 @@ def search_api( mime_type: Optional[str] = Query(None, description="Filter by MIME type (e.g. application/pdf)"), document_type: Optional[str] = Query(None, description="Filter by document type (e.g. Invoice)"), language: Optional[str] = Query(None, description="Filter by language code (e.g. de, en)"), + tags: Optional[str] = Query(None, description="Filter by tag (exact match)"), + sender: Optional[str] = Query(None, description="Filter by sender/absender (exact match)"), + text_quality: Optional[str] = Query( + None, + description="Filter by OCR text quality: no_text, low, medium, high", + ), date_from: Optional[int] = Query(None, description="Filter results created after this Unix timestamp"), date_to: Optional[int] = Query(None, description="Filter results created before this Unix timestamp"), page: int = Query(1, ge=1, description="Page number (1-based)"), @@ -50,6 +56,9 @@ def search_api( - mime_type: Filter by MIME type - document_type: Filter by document type - language: Filter by language code + - tags: Filter by tag (exact match on a single tag) + - sender: Filter by sender/absender (exact match) + - text_quality: Filter by OCR text quality (no_text, low, medium, high) - date_from: Unix timestamp lower bound - date_to: Unix timestamp upper bound - page: Page number (default: 1) @@ -57,7 +66,7 @@ def search_api( Example: ``` - GET /api/search?q=invoice&document_type=Invoice&date_from=1704067200&page=1&per_page=20 + GET /api/search?q=invoice&document_type=Invoice&tags=amazon&date_from=1704067200&page=1&per_page=20 ``` Response: @@ -90,6 +99,9 @@ def search_api( mime_type=mime_type, document_type=document_type, language=language, + tags=tags, + sender=sender, + text_quality=text_quality, date_from=date_from, date_to=date_to, page=page, diff --git a/app/utils/meilisearch_client.py b/app/utils/meilisearch_client.py index c99beb87..eccd2993 100644 --- a/app/utils/meilisearch_client.py +++ b/app/utils/meilisearch_client.py @@ -33,8 +33,10 @@ _INDEX_SETTINGS = { "document_type", "language", "tags", + "sender", "created_at_ts", "file_id", + "ocr_text_length", ], "sortableAttributes": [ "created_at_ts", @@ -55,6 +57,7 @@ _INDEX_SETTINGS = { "file_size", "created_at_ts", "ocr_text", + "ocr_text_length", ], "rankingRules": [ "words", @@ -142,6 +145,7 @@ def _build_document(file_record: "FileRecord", text: str, metadata: dict) -> dic "file_size": file_record.file_size or 0, "created_at_ts": created_at_ts, "ocr_text": text or "", + "ocr_text_length": len(text) if text else 0, } @@ -202,6 +206,9 @@ def search_documents( mime_type: Optional[str] = None, document_type: Optional[str] = None, language: Optional[str] = None, + tags: Optional[str] = None, + sender: Optional[str] = None, + text_quality: Optional[str] = None, date_from: Optional[int] = None, date_to: Optional[int] = None, page: int = 1, @@ -214,6 +221,9 @@ def search_documents( mime_type: Optional MIME-type filter. document_type: Optional document type filter. language: Optional language filter (ISO 639-1, e.g. "de"). + tags: Optional tag filter (exact match on a single tag). + sender: Optional sender/absender filter (exact match). + text_quality: Optional text quality filter: no_text, low, medium, high. date_from: Optional lower bound Unix timestamp for created_at. date_to: Optional upper bound Unix timestamp for created_at. page: 1-based page number. @@ -240,6 +250,21 @@ def search_documents( filters.append(f'document_type = "{document_type}"') if language: filters.append(f'language = "{language}"') + if tags: + filters.append(f'tags = "{tags}"') + if sender: + filters.append(f'sender = "{sender}"') + if text_quality: + # Translate text_quality labels into ocr_text_length ranges + _tq_filters = { + "no_text": "ocr_text_length = 0", + "low": "ocr_text_length > 0 AND ocr_text_length < 500", + "medium": "ocr_text_length >= 500 AND ocr_text_length < 2000", + "high": "ocr_text_length >= 2000", + } + tq_expr = _tq_filters.get(text_quality) + if tq_expr: + filters.append(tq_expr) if date_from is not None: filters.append(f"created_at_ts >= {date_from}") if date_to is not None: diff --git a/docs/API.md b/docs/API.md index b045bc7c..a877dece 100644 --- a/docs/API.md +++ b/docs/API.md @@ -333,10 +333,62 @@ GET /api/files?status=completed&mime_type=application/pdf&tags=invoice&date_from > **Tip**: Filter state is reflected in query parameters, making URLs shareable as bookmarks or direct links. +### Full-Text Search + +**GET** `/api/search` + +Search documents by full text across OCR content, titles, filenames, tags, sender, and document type. Powered by Meilisearch. + +**Query Parameters**: + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `q` | string | Yes | Full-text search query (1–512 chars) | +| `mime_type` | string | No | Filter by MIME type (e.g. `application/pdf`) | +| `document_type` | string | No | Filter by document type (e.g. `Invoice`) | +| `language` | string | No | Filter by language code (e.g. `de`, `en`) | +| `tags` | string | No | Filter by tag (exact match on a single tag) | +| `sender` | string | No | Filter by sender/absender (exact match) | +| `text_quality` | string | No | Filter by OCR text quality: `no_text`, `low`, `medium`, `high` | +| `date_from` | int | No | Filter results created after this Unix timestamp | +| `date_to` | int | No | Filter results created before this Unix timestamp | +| `page` | int | No | Page number, default: 1 | +| `per_page` | int | No | Results per page (1–100), default: 20 | + +**Example**: +``` +GET /api/search?q=invoice&document_type=Invoice&tags=amazon&text_quality=high&page=1 +``` + +**Response**: +```json +{ + "results": [ + { + "file_id": 42, + "original_filename": "2026-01-15_Invoice_Amazon.pdf", + "document_title": "Amazon Invoice January 2026", + "document_type": "Invoice", + "tags": ["amazon", "invoice"], + "_formatted": { + "document_title": "Amazon Invoice January 2026", + "ocr_text": "...total amount of the invoice is..." + } + } + ], + "total": 42, + "page": 1, + "pages": 3, + "query": "invoice" +} +``` + ### Saved Searches Saved searches allow users to save and reuse filter combinations. Each user can store up to 50 saved searches. +Saved searches are used on both the **Files** page (for file management filters) and the **Search** page (for content-finding filters including full-text queries). + #### List Saved Searches **GET** `/api/saved-searches` @@ -350,8 +402,9 @@ Returns all saved searches for the current user. "id": 1, "name": "Recent Invoices", "filters": { + "q": "invoice total", "tags": "invoice", - "status": "completed", + "document_type": "Invoice", "date_from": "2026-01-01" }, "created_at": "2026-03-01T10:00:00Z", @@ -369,14 +422,21 @@ Returns all saved searches for the current user. { "name": "Recent Invoices", "filters": { + "q": "invoice total", "tags": "invoice", - "status": "completed", + "document_type": "Invoice", "date_from": "2026-01-01" } } ``` -**Allowed filter keys**: `search`, `mime_type`, `status`, `date_from`, `date_to`, `storage_provider`, `tags`, `sort_by`, `sort_order` +**Allowed filter keys**: + +Files-view keys: `search`, `mime_type`, `status`, `storage_provider`, `sort_by`, `sort_order` + +Search-view keys: `q`, `document_type`, `language`, `sender`, `text_quality` + +Shared keys: `tags`, `date_from`, `date_to` **Response** (201 Created): The created saved search object. diff --git a/docs/UserGuide.md b/docs/UserGuide.md index 57a4d56d..98ff365c 100644 --- a/docs/UserGuide.md +++ b/docs/UserGuide.md @@ -119,16 +119,35 @@ The **Files** page includes a full-text search bar (labelled "Full-Text Search") ### Dedicated Search Page -For a more focused search experience, use the **Search** page accessible from the main navigation: +For a more focused content-finding experience, use the **Search** page accessible from the main navigation: 1. Navigate to the **Search** page 2. Type your query into the search box — results appear automatically as you type (or press Enter) -3. Results are displayed in a Google-style format showing: +3. Use the **content-finding filters** to narrow results: + - **Document Type** — e.g. Invoice, Contract + - **Tags** — filter by a specific tag + - **Sender** — filter by sender / absender + - **Language** — filter by ISO language code (e.g. `de`, `en`) + - **Text Quality** — filter by OCR text quality (High, Medium, Low, No text) + - **Date From / Date To** — restrict results to a date range +4. Results are displayed in a Google-style format showing: - **Document title** (linked to the file detail page) - **Filename** - **Document type**, **sender**, and **tag** badges - **Content preview** with highlighted matching terms -4. Use pagination to browse through large result sets +5. Use pagination to browse through large result sets + +### Saved Searches + +Both the **Files** and **Search** pages support **saved searches** — named filter presets you can create and reuse: + +1. Apply your desired filters (and optionally a search query on the Search page) +2. Click **Save Current** in the saved searches bar +3. Enter a name for the saved search +4. Your saved search appears as a clickable tag — click it to instantly re-apply those filters +5. Click the **×** button next to a saved search to delete it + +On the **Search** page, saved searches store the full-text query (`q`) along with all active content-finding filters. On the **Files** page, saved searches store the file management filters (filename search, MIME type, status, etc.). The search page is also accessible via URL with a pre-filled query: `/search?q=invoice` diff --git a/frontend/templates/search.html b/frontend/templates/search.html index e78b28e8..abac2a8a 100644 --- a/frontend/templates/search.html +++ b/frontend/templates/search.html @@ -7,7 +7,7 @@ .search-container { max-width: 800px; margin: 0 auto; } .search-box { display: flex; gap: 0.5rem; align-items: center; - margin-bottom: 1.5rem; + margin-bottom: 0.75rem; } .search-box input { flex: 1; padding: 0.75rem 1rem; @@ -24,11 +24,48 @@ white-space: nowrap; } .search-box button:hover { background-color: #2563eb; } + .search-filters { + display: flex; flex-wrap: wrap; gap: 0.5rem; align-items: flex-end; + margin-bottom: 0.75rem; padding: 0.75rem; + border: 1px solid #e5e7eb; border-radius: 0.5rem; + background: #f9fafb; + } + .search-filter-item { display: flex; flex-direction: column; gap: 0.2rem; } + .search-filter-item label { + font-size: 0.75rem; font-weight: 600; color: #4b5563; + } + .search-filter-item select, + .search-filter-item input { + padding: 0.4rem 0.5rem; border: 1px solid #d1d5db; + border-radius: 0.375rem; font-size: 0.8rem; min-width: 120px; + } + .search-saved-bar { + display: flex; align-items: center; gap: 0.5rem; flex-wrap: wrap; + margin-bottom: 1rem; font-size: 0.85rem; + } + .search-saved-bar .saved-label { + font-weight: 600; white-space: nowrap; + } + .saved-search-tag { + display: inline-flex; align-items: center; gap: 0.25rem; + background: #e5e7eb; padding: 0.2rem 0.5rem; border-radius: 0.25rem; + font-size: 0.8rem; + } + .saved-search-tag a { text-decoration: none; color: inherit; } + .saved-search-tag button { + border: none; background: none; cursor: pointer; color: #6b7280; + padding: 0; line-height: 1; + } + .btn-save-search { + margin-left: auto; font-size: 0.8rem; padding: 0.25rem 0.5rem; + cursor: pointer; border: 1px solid #d1d5db; border-radius: 0.25rem; + background: white; color: #374151; white-space: nowrap; + } + .btn-save-search:hover { background: #f3f4f6; } .search-summary { font-size: 0.875rem; color: #6b7280; margin-bottom: 1rem; } - /* Google-style result cards */ .search-result { margin-bottom: 1.5rem; } @@ -101,6 +138,60 @@ + +
+
+ + +
+
+ + +
+
+ + +
+
+ + +
+
+ + +
+
+ + +
+
+ + +
+
+ +
+
+ + +
+ Saved Searches: + + Loading... + + +
+ @@ -119,10 +210,50 @@ const summaryDiv = document.getElementById('search-summary'); const paginationDiv = document.getElementById('search-pagination'); + // Filter elements + const filterDocType = document.getElementById('filter-document-type'); + const filterTags = document.getElementById('filter-tags'); + const filterSender = document.getElementById('filter-sender'); + const filterLanguage = document.getElementById('filter-language'); + const filterTextQuality = document.getElementById('filter-text-quality'); + const filterDateFrom = document.getElementById('filter-date-from'); + const filterDateTo = document.getElementById('filter-date-to'); + let _debounce = null; let _currentPage = 1; const PER_PAGE = 20; + /** Collect active filter values into an object. */ + function getActiveFilters() { + const filters = {}; + const dt = filterDocType.value.trim(); + if (dt) filters.document_type = dt; + const tg = filterTags.value.trim(); + if (tg) filters.tags = tg; + const sn = filterSender.value.trim(); + if (sn) filters.sender = sn; + const ln = filterLanguage.value.trim(); + if (ln) filters.language = ln; + const tq = filterTextQuality.value; + if (tq) filters.text_quality = tq; + const df = filterDateFrom.value; + if (df) filters.date_from = Math.floor(new Date(df + 'T00:00:00Z').getTime() / 1000); + const dTo = filterDateTo.value; + if (dTo) filters.date_to = Math.floor(new Date(dTo + 'T23:59:59Z').getTime() / 1000); + return filters; + } + + /** Apply filter values from an object to the UI inputs. */ + function applyFiltersToUI(filters) { + filterDocType.value = filters.document_type || ''; + filterTags.value = filters.tags || ''; + filterSender.value = filters.sender || ''; + filterLanguage.value = filters.language || ''; + filterTextQuality.value = filters.text_quality || ''; + filterDateFrom.value = filters.date_from || ''; + filterDateTo.value = filters.date_to || ''; + } + function doSearch(page) { const q = searchInput.value.trim(); if (!q) { @@ -136,6 +267,15 @@ // Update URL without reload const url = new URL(window.location); url.searchParams.set('q', q); + // Sync filter params to URL + const activeFilters = getActiveFilters(); + Object.keys(activeFilters).forEach(function(key) { + url.searchParams.set(key, activeFilters[key]); + }); + // Remove filter keys not in activeFilters + ['document_type', 'tags', 'sender', 'language', 'text_quality', 'date_from', 'date_to'].forEach(function(key) { + if (!activeFilters[key]) url.searchParams.delete(key); + }); window.history.replaceState({}, '', url); // Loading indicator @@ -143,11 +283,16 @@ summaryDiv.style.display = 'none'; paginationDiv.style.display = 'none'; - const params = new URLSearchParams({ q, page: _currentPage, per_page: PER_PAGE }); + const params = new URLSearchParams({ q: q, page: _currentPage, per_page: PER_PAGE }); + // Append filter params (date_from/date_to already converted to timestamps) + Object.keys(activeFilters).forEach(function(key) { + params.set(key, activeFilters[key]); + }); + fetch('/api/search?' + params.toString()) - .then(r => { if (!r.ok) throw new Error('Search returned ' + r.status); return r.json(); }) - .then(data => renderResults(data, q)) - .catch(err => { + .then(function(r) { if (!r.ok) throw new Error('Search returned ' + r.status); return r.json(); }) + .then(function(data) { renderResults(data, q); }) + .catch(function(err) { resultsDiv.innerHTML = '

Search is temporarily unavailable. Please try again in a moment.

' + escapeHtml(err.message) + '

'; }); } @@ -163,13 +308,10 @@ * escape everything else to prevent XSS from indexed content. */ function sanitizeHighlight(html) { - // Temporarily replace and with placeholders - var safe = html + let safe = html .replace(//gi, '\x00MARK_OPEN\x00') .replace(/<\/mark>/gi, '\x00MARK_CLOSE\x00'); - // Escape all remaining HTML safe = escapeHtml(safe); - // Restore the tags safe = safe .replace(/\x00MARK_OPEN\x00/g, '') .replace(/\x00MARK_CLOSE\x00/g, ''); @@ -177,43 +319,42 @@ } function renderResults(data, q) { - const { results, total, page, pages } = data; + const results = data.results; + const total = data.total; + const page = data.page; + const pages = data.pages; - // Summary summaryDiv.textContent = total + ' result' + (total !== 1 ? 's' : '') + ' for "' + q + '"'; summaryDiv.style.display = 'block'; if (!results || results.length === 0) { - resultsDiv.innerHTML = '

No documents found matching your query.

'; + resultsDiv.innerHTML = '

No documents found matching your query.

'; paginationDiv.style.display = 'none'; return; } resultsDiv.innerHTML = results.map(function(hit) { - var fmt = hit._formatted || {}; - var title = fmt.document_title || hit.document_title || hit.original_filename || '(untitled)'; - var filename = hit.original_filename || ''; - var snippet = fmt.ocr_text || ''; - var tags = Array.isArray(hit.tags) ? hit.tags : (hit.tags ? [hit.tags] : []); - var docType = hit.document_type || ''; - var sender = hit.sender || hit.absender || ''; - var fileUrl = '/files/' + hit.file_id; + const fmt = hit._formatted || {}; + const title = fmt.document_title || hit.document_title || hit.original_filename || '(untitled)'; + const filename = hit.original_filename || ''; + const snippet = fmt.ocr_text || ''; + const tags = Array.isArray(hit.tags) ? hit.tags : (hit.tags ? [hit.tags] : []); + const docType = hit.document_type || ''; + const sender = hit.sender || hit.absender || ''; + const fileUrl = '/files/' + hit.file_id; - // Build badges - var badges = ''; + let badges = ''; if (docType) badges += '' + escapeHtml(docType) + ''; - if (sender) badges += ' ' + escapeHtml(sender) + ''; + if (sender) badges += ' ' + escapeHtml(sender) + ''; tags.forEach(function(t) { badges += '' + escapeHtml(t) + ''; }); - // Snippet: use highlighted text, truncate if very long - var snippetHtml = ''; + let snippetHtml = ''; if (snippet) { - var trimmed = snippet.length > 500 ? snippet.substring(0, 500) + '…' : snippet; + const trimmed = snippet.length > 500 ? snippet.substring(0, 500) + '…' : snippet; snippetHtml = '
…' + sanitizeHighlight(trimmed) + '…
'; } - // Sanitize title (may contain highlights from _formatted) - var safeTitle = (fmt.document_title) ? sanitizeHighlight(title) : escapeHtml(title); + const safeTitle = (fmt.document_title) ? sanitizeHighlight(title) : escapeHtml(title); return '
' + '' + @@ -223,9 +364,8 @@ '
'; }).join(''); - // Pagination if (pages > 1) { - var btns = []; + const btns = []; if (page > 1) btns.push(''); btns.push('Page ' + page + ' of ' + pages + ''); if (page < pages) btns.push(''); @@ -238,18 +378,18 @@ // Event delegation for pagination buttons paginationDiv.addEventListener('click', function(e) { - var btn = e.target.closest('button[data-page]'); + const btn = e.target.closest('button[data-page]'); if (btn) doSearch(parseInt(btn.getAttribute('data-page'), 10)); }); - // Event listeners + // Event listeners for search searchBtn.addEventListener('click', function() { doSearch(1); }); searchInput.addEventListener('keydown', function(e) { if (e.key === 'Enter') { doSearch(1); } }); searchInput.addEventListener('input', function() { clearTimeout(_debounce); - var val = searchInput.value.trim(); + const val = searchInput.value.trim(); if (val.length < 2) { resultsDiv.innerHTML = ''; summaryDiv.style.display = 'none'; @@ -259,9 +399,104 @@ _debounce = setTimeout(function() { doSearch(1); }, 400); }); - // If q was provided via URL, search immediately - if (searchInput.value.trim().length >= 2) { - doSearch(1); + // Clear filters button + document.getElementById('clear-filters-btn').addEventListener('click', function() { + applyFiltersToUI({}); + if (searchInput.value.trim().length >= 2) doSearch(1); + }); + + // Re-run search when filters change + const filterInputs = [filterDocType, filterTags, filterSender, filterLanguage, filterTextQuality, filterDateFrom, filterDateTo]; + filterInputs.forEach(function(el) { + el.addEventListener('change', function() { + if (searchInput.value.trim().length >= 2) doSearch(1); + }); + }); + + // ---- Saved searches functionality ---- + function loadSavedSearches() { + fetch('/api/saved-searches') + .then(function(response) { return response.json(); }) + .then(function(searches) { + const container = document.getElementById('saved-searches-list'); + if (!container) return; + if (!searches || searches.length === 0) { + container.innerHTML = 'No saved searches yet'; + return; + } + container.innerHTML = searches.map(function(s) { + const params = new URLSearchParams(s.filters); + return '' + + '' + escapeHtml(s.name) + '' + + '' + + ''; + }).join(''); + }) + .catch(function() { + const container = document.getElementById('saved-searches-list'); + if (container) container.innerHTML = 'Could not load saved searches'; + }); } + + document.getElementById('save-search-btn').addEventListener('click', function() { + const filters = getActiveFilters(); + const q = searchInput.value.trim(); + // For saved searches on this page, store date_from/date_to as ISO strings + // instead of timestamps so the URL params are human-readable. + if (filterDateFrom.value) filters.date_from = filterDateFrom.value; + if (filterDateTo.value) filters.date_to = filterDateTo.value; + if (q) filters.q = q; + + if (Object.keys(filters).length === 0) { + alert('No search query or filters to save. Please enter a query or set at least one filter.'); + return; + } + const name = prompt('Enter a name for this saved search:'); + if (!name || !name.trim()) return; + fetch('/api/saved-searches', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ name: name.trim(), filters: filters }) + }) + .then(function(response) { + if (!response.ok) return response.json().then(function(d) { throw new Error(d.detail || 'Failed to save'); }); + return response.json(); + }) + .then(function() { loadSavedSearches(); }) + .catch(function(error) { alert(error.message); }); + }); + + function deleteSavedSearch(id) { + if (!confirm('Delete this saved search?')) return; + fetch('/api/saved-searches/' + id, { method: 'DELETE' }) + .then(function(response) { + if (!response.ok) throw new Error('Failed to delete'); + loadSavedSearches(); + }) + .catch(function(error) { alert(error.message); }); + } + + // ---- Initialize from URL params ---- + (function() { + const urlParams = new URLSearchParams(window.location.search); + // Populate filters from URL params (for saved search links) + if (urlParams.get('document_type')) filterDocType.value = urlParams.get('document_type'); + if (urlParams.get('tags')) filterTags.value = urlParams.get('tags'); + if (urlParams.get('sender')) filterSender.value = urlParams.get('sender'); + if (urlParams.get('language')) filterLanguage.value = urlParams.get('language'); + if (urlParams.get('text_quality')) filterTextQuality.value = urlParams.get('text_quality'); + if (urlParams.get('date_from')) filterDateFrom.value = urlParams.get('date_from'); + if (urlParams.get('date_to')) filterDateTo.value = urlParams.get('date_to'); + + // Load saved searches + loadSavedSearches(); + + // If q was provided via URL, search immediately + if (searchInput.value.trim().length >= 2) { + doSearch(1); + } + })(); {% endblock %} diff --git a/tests/test_api_advanced_filters.py b/tests/test_api_advanced_filters.py index 514d99cd..03bd7b24 100644 --- a/tests/test_api_advanced_filters.py +++ b/tests/test_api_advanced_filters.py @@ -348,3 +348,36 @@ class TestSavedSearchesCRUD: assert len(data["filters"]) == 8 assert data["filters"]["search"] == "invoice" assert data["filters"]["tags"] == "invoice,amazon" + + def test_saved_search_with_fulltext_query(self, client: TestClient): + """Saved search can include full-text query (q) for the search view.""" + payload = { + "name": "Invoice Search", + "filters": {"q": "invoice total amount", "document_type": "Invoice"}, + } + response = client.post("/api/saved-searches", json=payload) + assert response.status_code == 201 + data = response.json() + assert data["filters"]["q"] == "invoice total amount" + assert data["filters"]["document_type"] == "Invoice" + + def test_saved_search_content_finding_filters(self, client: TestClient): + """Saved search accepts content-finding filter keys (language, sender, text_quality).""" + payload = { + "name": "German Invoices", + "filters": { + "q": "rechnung", + "language": "de", + "sender": "ACME GmbH", + "text_quality": "high", + "tags": "invoice", + }, + } + response = client.post("/api/saved-searches", json=payload) + assert response.status_code == 201 + data = response.json() + assert data["filters"]["q"] == "rechnung" + assert data["filters"]["language"] == "de" + assert data["filters"]["sender"] == "ACME GmbH" + assert data["filters"]["text_quality"] == "high" + assert data["filters"]["tags"] == "invoice" diff --git a/tests/test_api_search.py b/tests/test_api_search.py index c44a4db6..43322cb8 100644 --- a/tests/test_api_search.py +++ b/tests/test_api_search.py @@ -97,6 +97,7 @@ class TestMeilisearchIndexDocument: assert doc["file_id"] == 1 assert doc["document_title"] == "Invoice January 2026" assert "invoice" in doc["tags"] + assert doc["ocr_text_length"] == len("This is an invoice for services rendered") def test_index_document_meilisearch_error(self): """index_document returns False on Meilisearch exception.""" @@ -179,6 +180,62 @@ class TestMeilisearchSearchDocuments: assert 'mime_type = "application/pdf"' in search_params["filter"] assert 'language = "de"' in search_params["filter"] + def test_search_with_tags_filter(self): + """search_documents passes tags filter to Meilisearch.""" + mock_client, mock_index = self._make_mock_client(hits=[], total=0) + + with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client): + from app.utils.meilisearch_client import search_documents + + search_documents("test", tags="invoice", page=1, per_page=10) + + call_kwargs = mock_index.search.call_args + search_params = call_kwargs[0][1] + assert "filter" in search_params + assert 'tags = "invoice"' in search_params["filter"] + + def test_search_with_sender_filter(self): + """search_documents passes sender filter to Meilisearch.""" + mock_client, mock_index = self._make_mock_client(hits=[], total=0) + + with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client): + from app.utils.meilisearch_client import search_documents + + search_documents("test", sender="ACME Corp", page=1, per_page=10) + + call_kwargs = mock_index.search.call_args + search_params = call_kwargs[0][1] + assert "filter" in search_params + assert 'sender = "ACME Corp"' in search_params["filter"] + + def test_search_with_text_quality_high(self): + """search_documents translates text_quality=high to ocr_text_length filter.""" + mock_client, mock_index = self._make_mock_client(hits=[], total=0) + + with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client): + from app.utils.meilisearch_client import search_documents + + search_documents("test", text_quality="high", page=1, per_page=10) + + call_kwargs = mock_index.search.call_args + search_params = call_kwargs[0][1] + assert "filter" in search_params + assert "ocr_text_length >= 2000" in search_params["filter"] + + def test_search_with_text_quality_no_text(self): + """search_documents translates text_quality=no_text to ocr_text_length filter.""" + mock_client, mock_index = self._make_mock_client(hits=[], total=0) + + with patch("app.utils.meilisearch_client.get_meilisearch_client", return_value=mock_client): + from app.utils.meilisearch_client import search_documents + + search_documents("test", text_quality="no_text", page=1, per_page=10) + + call_kwargs = mock_index.search.call_args + search_params = call_kwargs[0][1] + assert "filter" in search_params + assert "ocr_text_length = 0" in search_params["filter"] + def test_search_pagination(self): """search_documents applies correct offset for page 2.""" mock_client, mock_index = self._make_mock_client(hits=[], total=50) @@ -271,6 +328,9 @@ class TestSearchAPIEndpoint: mime_type="application/pdf", document_type=None, language="en", + tags=None, + sender=None, + text_quality=None, date_from=None, date_to=None, page=2, @@ -294,6 +354,9 @@ class TestSearchAPIEndpoint: mime_type=None, document_type=None, language=None, + tags=None, + sender=None, + text_quality=None, date_from=None, date_to=None, page=1, @@ -310,3 +373,30 @@ class TestSearchAPIEndpoint: call_kwargs = mock_search.call_args assert call_kwargs[1]["date_from"] == 1704067200 assert call_kwargs[1]["date_to"] == 1735689600 + + def test_search_endpoint_tags_filter(self, client): + """GET /api/search?q=...&tags=invoice passes tags to search_documents.""" + mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "test"} + with patch("app.api.search.search_documents", return_value=mock_result) as mock_search: + response = client.get("/api/search?q=test&tags=invoice") + + assert response.status_code == 200 + assert mock_search.call_args[1]["tags"] == "invoice" + + def test_search_endpoint_sender_filter(self, client): + """GET /api/search?q=...&sender=ACME passes sender to search_documents.""" + mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "test"} + with patch("app.api.search.search_documents", return_value=mock_result) as mock_search: + response = client.get("/api/search?q=test&sender=ACME") + + assert response.status_code == 200 + assert mock_search.call_args[1]["sender"] == "ACME" + + def test_search_endpoint_text_quality_filter(self, client): + """GET /api/search?q=...&text_quality=high passes text_quality to search_documents.""" + mock_result = {"results": [], "total": 0, "page": 1, "pages": 0, "query": "test"} + with patch("app.api.search.search_documents", return_value=mock_result) as mock_search: + response = client.get("/api/search?q=test&text_quality=high") + + assert response.status_code == 200 + assert mock_search.call_args[1]["text_quality"] == "high" diff --git a/tests/test_views_search.py b/tests/test_views_search.py index b3d6bbe3..788ec4c8 100644 --- a/tests/test_views_search.py +++ b/tests/test_views_search.py @@ -39,3 +39,33 @@ class TestSearchPage: long_query = "a" * 600 response = client.get(f"/search?q={long_query}") assert response.status_code == 422 + + def test_search_page_contains_filter_elements(self, client): + """GET /search contains content-finding filter UI elements.""" + response = client.get("/search") + assert response.status_code == 200 + assert 'id="filter-document-type"' in response.text + assert 'id="filter-tags"' in response.text + assert 'id="filter-sender"' in response.text + assert 'id="filter-language"' in response.text + assert 'id="filter-text-quality"' in response.text + assert 'id="filter-date-from"' in response.text + assert 'id="filter-date-to"' in response.text + + def test_search_page_contains_saved_searches(self, client): + """GET /search contains saved searches UI elements.""" + response = client.get("/search") + assert response.status_code == 200 + assert 'id="saved-searches-list"' in response.text + assert 'id="save-search-btn"' in response.text + assert "/api/saved-searches" in response.text + + def test_search_page_text_quality_options(self, client): + """GET /search contains text quality filter with expected options.""" + response = client.get("/search") + assert response.status_code == 200 + text = response.text + assert 'value="high"' in text + assert 'value="medium"' in text + assert 'value="low"' in text + assert 'value="no_text"' in text