diff --git a/.env.demo b/.env.demo index 12f9ebf8..a54995f2 100644 --- a/.env.demo +++ b/.env.demo @@ -299,3 +299,14 @@ NOTIFY_ON_FILE_PROCESSED=True # Uptime Kuma UPTIME_KUMA_URL=https://status.example.com/api/push/abcdef123456?status=up UPTIME_KUMA_PING_INTERVAL=5 + +# **Full-Text Search (Meilisearch)** +# URL for the Meilisearch instance. +# Default is "http://meilisearch:7700" — the Docker Compose / K8s service name — +# so container-to-container networking works without extra configuration. +# Override to "http://localhost:7700" only when running the API process outside Docker. +MEILISEARCH_URL=http://meilisearch:7700 +# Optional master/API key for secured Meilisearch instances +# MEILISEARCH_API_KEY=your_master_key_here +MEILISEARCH_INDEX_NAME=documents +ENABLE_SEARCH=True diff --git a/app/api/__init__.py b/app/api/__init__.py index 20bafc6c..b483248a 100644 --- a/app/api/__init__.py +++ b/app/api/__init__.py @@ -15,6 +15,7 @@ from app.api.logs import router as logs_router from app.api.onedrive import router as onedrive_router from app.api.openai import router as openai_router from app.api.process import router as process_router +from app.api.search import router as search_router from app.api.settings import router as settings_router from app.api.url_upload import router as url_upload_router @@ -40,3 +41,4 @@ router.include_router(google_drive_router) router.include_router(logs_router) router.include_router(settings_router) router.include_router(url_upload_router) +router.include_router(search_router) diff --git a/app/api/search.py b/app/api/search.py new file mode 100644 index 00000000..2b6a5117 --- /dev/null +++ b/app/api/search.py @@ -0,0 +1,99 @@ +"""Full-text search API endpoints. + +Provides document search across OCR text, AI metadata, filenames, and tags +via Meilisearch. Designed to serve as the backend for the UI search bar on +the /files page and as a standalone API for integrations. + +Future extension point: the OCR text stored in the index is also suitable +for RAG (Retrieval Augmented Generation) chatbot workflows. +""" + +import logging +from typing import Optional + +from fastapi import APIRouter, Query, Request + +from app.auth import require_login +from app.utils.meilisearch_client import search_documents + +logger = logging.getLogger(__name__) + +router = APIRouter() + + +@router.get("/search") +@require_login +def search_api( + request: Request, + q: str = Query(..., min_length=1, max_length=512, description="Full-text search query"), + mime_type: Optional[str] = Query(None, description="Filter by MIME type (e.g. application/pdf)"), + document_type: Optional[str] = Query(None, description="Filter by document type (e.g. Invoice)"), + language: Optional[str] = Query(None, description="Filter by language code (e.g. de, en)"), + date_from: Optional[int] = Query(None, description="Filter results created after this Unix timestamp"), + date_to: Optional[int] = Query(None, description="Filter results created before this Unix timestamp"), + page: int = Query(1, ge=1, description="Page number (1-based)"), + per_page: int = Query(20, ge=1, le=100, description="Results per page"), +): + """Search documents by full text, metadata, and tags. + + Searches across: + - Document title and filename + - OCR / extracted text + - Tags, sender, recipient, document type + - Correspondent and reference number + + Results are ranked by Meilisearch relevance and include highlighted + snippets showing where the query terms matched. + + Query Parameters: + - q: Search query (required) + - mime_type: Filter by MIME type + - document_type: Filter by document type + - language: Filter by language code + - date_from: Unix timestamp lower bound + - date_to: Unix timestamp upper bound + - page: Page number (default: 1) + - per_page: Results per page (default: 20, max: 100) + + Example: + ``` + GET /api/search?q=invoice&document_type=Invoice&date_from=1704067200&page=1&per_page=20 + ``` + + Response: + ```json + { + "results": [ + { + "file_id": 42, + "original_filename": "2026-01-15_Invoice_Amazon.pdf", + "document_title": "Amazon Invoice January 2026", + "document_type": "Invoice", + "tags": ["amazon", "invoice"], + "_formatted": { + "document_title": "Amazon Invoice January 2026", + "ocr_text": "...total amount of the invoice is..." + } + } + ], + "total": 42, + "page": 1, + "pages": 3, + "query": "invoice" + } + ``` + """ + logger.info(f"Search request: q={q!r}, mime_type={mime_type}, page={page}, per_page={per_page}") + + result = search_documents( + q, + mime_type=mime_type, + document_type=document_type, + language=language, + date_from=date_from, + date_to=date_to, + page=page, + per_page=per_page, + ) + + return result diff --git a/app/config.py b/app/config.py index 3bbae76a..41398e91 100644 --- a/app/config.py +++ b/app/config.py @@ -207,6 +207,15 @@ class Settings(BaseSettings): uptime_kuma_url: Optional[str] = None uptime_kuma_ping_interval: int = 5 # Default ping interval in minutes + # Meilisearch settings (full-text search engine) + # Default uses the Docker Compose / K8s service name so container-to-container + # networking works without any extra configuration. Override to + # "http://localhost:7700" only when running the API process outside of Docker. + meilisearch_url: str = "http://meilisearch:7700" + meilisearch_api_key: Optional[str] = None # Master or API key (optional for local dev) + meilisearch_index_name: str = "documents" + enable_search: bool = True # Enable Meilisearch full-text search integration + # HTTP request settings http_request_timeout: int = 120 # Default timeout for HTTP requests in seconds (handles large file operations) diff --git a/app/models.py b/app/models.py index 78c09df9..8997bc3c 100644 --- a/app/models.py +++ b/app/models.py @@ -55,6 +55,15 @@ class FileRecord(Base): # If this is a duplicate, record the ID of the original file for reference duplicate_of_id = Column(Integer, ForeignKey(_FILES_ID_FK), nullable=True) + # Full OCR/extracted text for full-text search and RAG + ocr_text = Column(Text, nullable=True) + + # AI-extracted metadata stored as JSON string (filename, tags, title, sender, etc.) + ai_metadata = Column(Text, nullable=True) + + # Human-readable document title from AI metadata + document_title = Column(String, nullable=True) + # Timestamp when we inserted this record created_at = Column(DateTime(timezone=True), server_default=func.now()) diff --git a/app/tasks/embed_metadata_into_pdf.py b/app/tasks/embed_metadata_into_pdf.py index 82bdb8bb..3bc78664 100644 --- a/app/tasks/embed_metadata_into_pdf.py +++ b/app/tasks/embed_metadata_into_pdf.py @@ -195,8 +195,26 @@ def embed_metadata_into_pdf(self, local_file_path: str, extracted_text: str, met original_file_path = file_record.original_file_path # Update the processed_file_path in the database file_record.processed_file_path = final_file_path + # Persist extracted text and AI metadata to DB for full-text search / RAG + file_record.ocr_text = extracted_text or None + if metadata: + try: + file_record.ai_metadata = json.dumps(metadata, ensure_ascii=False) + except Exception as json_exc: + logger.warning(f"[{task_id}] Could not serialise ai_metadata: {json_exc}") + file_record.document_title = ( + metadata.get("title") or metadata.get("filename") or file_record.original_filename + ) db.commit() - logger.info(f"[{task_id}] Updated database with processed_file_path: {final_file_path}") + logger.info(f"[{task_id}] Updated database with processed_file_path and search fields") + + # Index into Meilisearch for full-text search (non-blocking, best-effort) + try: + from app.utils.meilisearch_client import index_document + + index_document(file_record, extracted_text or "", metadata or {}) + except Exception as search_exc: + logger.warning(f"[{task_id}] Meilisearch indexing failed (non-fatal): {search_exc}") # Persist the metadata into a JSON file with the same base name. # Include file path references for traceability diff --git a/app/utils/meilisearch_client.py b/app/utils/meilisearch_client.py new file mode 100644 index 00000000..288c699e --- /dev/null +++ b/app/utils/meilisearch_client.py @@ -0,0 +1,286 @@ +"""Meilisearch client utilities for full-text document search. + +This module provides functions for indexing documents into Meilisearch +and searching across OCR text, AI metadata, filenames, and tags. + +The search index is designed to support future RAG (Retrieval Augmented +Generation) workflows by storing full document text alongside structured +metadata fields. +""" + +import logging +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +# Index settings applied once at index creation / first use +_INDEX_SETTINGS = { + "searchableAttributes": [ + "document_title", + "original_filename", + "ocr_text", + "tags", + "sender", + "recipient", + "document_type", + "correspondent", + ], + "filterableAttributes": [ + "mime_type", + "document_type", + "language", + "tags", + "created_at_ts", + "file_id", + ], + "sortableAttributes": [ + "created_at_ts", + "file_size", + ], + "displayedAttributes": [ + "file_id", + "original_filename", + "document_title", + "document_type", + "tags", + "sender", + "recipient", + "correspondent", + "language", + "reference_number", + "mime_type", + "file_size", + "created_at_ts", + "ocr_text", + ], + "rankingRules": [ + "words", + "typo", + "proximity", + "attribute", + "sort", + "exactness", + ], +} + + +def get_meilisearch_client(): + """Return a configured Meilisearch client, or None if unavailable/disabled.""" + try: + import meilisearch + + from app.config import settings + + if not settings.enable_search: + return None + + kwargs: dict[str, Any] = {"url": settings.meilisearch_url} + if settings.meilisearch_api_key: + kwargs["api_key"] = settings.meilisearch_api_key + + client = meilisearch.Client(**kwargs) + return client + except ImportError: + logger.warning("meilisearch package not installed; search disabled") + return None + except Exception as exc: + logger.warning(f"Could not connect to Meilisearch: {exc}") + return None + + +def _get_or_create_index(client): + """Get the documents index, creating it with settings if it doesn't exist.""" + from app.config import settings + + index_name = settings.meilisearch_index_name + try: + index = client.get_index(index_name) + except Exception: + # Index doesn't exist – create it with file_id as primary key + task = client.create_index(index_name, {"primaryKey": "file_id"}) + client.wait_for_task(task.task_uid) + index = client.get_index(index_name) + # Apply search settings + try: + task = index.update_settings(_INDEX_SETTINGS) + client.wait_for_task(task.task_uid) + except Exception as exc: + logger.warning(f"Could not update Meilisearch index settings: {exc}") + return index + + +def _build_document(file_record, text: str, metadata: dict) -> dict: + """Build a Meilisearch document from a FileRecord and extracted content.""" + + tags = metadata.get("tags", []) + if isinstance(tags, str): + tags = [t.strip() for t in tags.split(",") if t.strip()] + + # Unix timestamp for sorting/filtering + created_at_ts = 0 + if file_record.created_at: + try: + created_at_ts = int(file_record.created_at.timestamp()) + except Exception as ts_exc: # noqa: BLE001 + logger.debug(f"Could not convert created_at to timestamp: {ts_exc}") + + return { + "file_id": file_record.id, + "original_filename": file_record.original_filename or "", + "document_title": metadata.get("title") or metadata.get("filename") or file_record.original_filename or "", + "document_type": metadata.get("document_type") or metadata.get("kommunikationsart") or "", + "tags": tags, + "sender": metadata.get("absender") or "", + "recipient": metadata.get("empfaenger") or "", + "correspondent": metadata.get("correspondent") or "", + "language": metadata.get("language") or "", + "reference_number": metadata.get("reference_number") or "", + "mime_type": file_record.mime_type or "", + "file_size": file_record.file_size or 0, + "created_at_ts": created_at_ts, + "ocr_text": text or "", + } + + +def index_document(file_record, text: str, metadata: dict) -> bool: + """Index a document in Meilisearch. + + Args: + file_record: FileRecord ORM instance with at minimum .id set. + text: Full OCR / extracted text for the document. + metadata: AI-extracted metadata dict. + + Returns: + True if indexing succeeded, False otherwise. + """ + client = get_meilisearch_client() + if client is None: + return False + + try: + index = _get_or_create_index(client) + doc = _build_document(file_record, text, metadata) + task = index.add_documents([doc]) + logger.info(f"Queued Meilisearch indexing for file_id={file_record.id} (task_uid={task.task_uid})") + return True + except Exception as exc: + logger.warning(f"Meilisearch indexing failed for file_id={file_record.id}: {exc}") + return False + + +def delete_document(file_id: int) -> bool: + """Remove a document from the Meilisearch index. + + Args: + file_id: The database ID of the file to remove. + + Returns: + True if deletion succeeded, False otherwise. + """ + client = get_meilisearch_client() + if client is None: + return False + + try: + from app.config import settings + + index = client.get_index(settings.meilisearch_index_name) + task = index.delete_document(file_id) + logger.info(f"Queued Meilisearch deletion for file_id={file_id} (task_uid={task.task_uid})") + return True + except Exception as exc: + logger.warning(f"Meilisearch deletion failed for file_id={file_id}: {exc}") + return False + + +def search_documents( + query: str, + *, + mime_type: Optional[str] = None, + document_type: Optional[str] = None, + language: Optional[str] = None, + date_from: Optional[int] = None, + date_to: Optional[int] = None, + page: int = 1, + per_page: int = 20, +) -> dict: + """Search documents in Meilisearch. + + Args: + query: Full-text search query string. + mime_type: Optional MIME-type filter. + document_type: Optional document type filter. + language: Optional language filter (ISO 639-1, e.g. "de"). + date_from: Optional lower bound Unix timestamp for created_at. + date_to: Optional upper bound Unix timestamp for created_at. + page: 1-based page number. + per_page: Results per page (max 100). + + Returns: + Dict with keys: results, total, page, pages, query. + Returns empty results dict on any error. + """ + empty: dict = {"results": [], "total": 0, "page": page, "pages": 0, "query": query} + + client = get_meilisearch_client() + if client is None: + return empty + + try: + index = _get_or_create_index(client) + + # Build filter expressions + filters: list[str] = [] + if mime_type: + filters.append(f'mime_type = "{mime_type}"') + if document_type: + filters.append(f'document_type = "{document_type}"') + if language: + filters.append(f'language = "{language}"') + if date_from is not None: + filters.append(f"created_at_ts >= {date_from}") + if date_to is not None: + filters.append(f"created_at_ts <= {date_to}") + + search_params: dict[str, Any] = { + "offset": (page - 1) * per_page, + "limit": per_page, + "attributesToHighlight": ["document_title", "original_filename", "ocr_text", "tags"], + "highlightPreTag": "", + "highlightPostTag": "", + "attributesToCrop": ["ocr_text"], + "cropLength": 200, + } + + if filters: + search_params["filter"] = " AND ".join(filters) + + result = index.search(query, search_params) + + hits = result.get("hits", []) + total = result.get("estimatedTotalHits", result.get("nbHits", len(hits))) + + # Attach highlights to each hit + formatted_results = [] + for hit in hits: + formatted = dict(hit) + # Include formatted (highlighted) snippets if available + if "_formatted" in hit: + formatted["_formatted"] = hit["_formatted"] + # Exclude raw ocr_text from results (use _formatted snippet instead) + formatted.pop("ocr_text", None) + formatted_results.append(formatted) + + pages = (total + per_page - 1) // per_page if total > 0 else 0 + + return { + "results": formatted_results, + "total": total, + "page": page, + "pages": pages, + "query": query, + } + + except Exception as exc: + logger.warning(f"Meilisearch search failed for query '{query}': {exc}") + return empty diff --git a/docker-compose.yaml b/docker-compose.yaml index e09c5b23..06e0f1c8 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -59,6 +59,15 @@ services: container_name: gotenberg restart: always + meilisearch: + image: getmeili/meilisearch:latest + container_name: document_meilisearch + restart: always + environment: + - MEILI_NO_ANALYTICS=true + volumes: + - /var/docparse/meilisearch:/meili_data + redis: image: redis:alpine container_name: document_redis diff --git a/docs/DeploymentGuide.md b/docs/DeploymentGuide.md index fb152e9c..cb66bfa9 100644 --- a/docs/DeploymentGuide.md +++ b/docs/DeploymentGuide.md @@ -1,40 +1,48 @@ # Deployment Guide -This guide provides instructions for deploying DocuElevate in various environments. +This guide covers all supported deployment methods for DocuElevate. + +## Table of Contents + +- [Prerequisites](#prerequisites) +- [Docker Compose Deployment](#docker-compose-deployment) *(recommended for single-server)* +- [Kubernetes / Helm Deployment](#kubernetes--helm-deployment) *(recommended for production scale-out)* +- [Production Considerations](#production-considerations) +- [Scaling](#scaling) +- [Backup Procedures](#backup-procedures) +- [Updates](#updates) +- [Troubleshooting](#troubleshooting) ## Prerequisites -- Docker and Docker Compose +- Docker and Docker Compose **or** a Kubernetes cluster with Helm 3 - Access to required external services (if configured): - AI provider API key (OpenAI, Anthropic, Gemini, or other configured provider) - Azure Document Intelligence - - Dropbox API - - Nextcloud instance - - Paperless NGX instance - - SMTP server (for email notifications) - - IMAP server(s) (for email attachment processing) - - Notification services (Discord, Telegram, etc. for system alerts) + - Dropbox, Google Drive, OneDrive, S3, or other storage APIs + - SMTP / IMAP server (for email processing) + - Notification services (Discord, Telegram, etc.) -## Docker Deployment +--- -Docker is the recommended deployment method for DocuElevate. +## Docker Compose Deployment + +Docker Compose is the quickest way to run DocuElevate on a single server. ### Step 1: Clone the Repository ```bash -git clone https://github.com/christianlouis/document-processor.git -cd document-processor +git clone https://github.com/christianlouis/DocuElevate.git +cd DocuElevate ``` ### Step 2: Configure Environment Variables -Create a `.env` file based on the example: - ```bash -cp .env.example .env +cp .env.demo .env ``` -Edit the `.env` file with your configuration settings. See the [Configuration Guide](ConfigurationGuide.md) for details. +Edit `.env` with your settings. See the [Configuration Guide](ConfigurationGuide.md) for all options. ### Step 3: Run with Docker Compose @@ -42,41 +50,237 @@ Edit the `.env` file with your configuration settings. See the [Configuration Gu docker-compose up -d ``` -This will start: -- The DocuElevate API server -- A worker for background tasks -- Redis for message broker and result storage -- Gotenberg for PDF processing +This starts: + +| Service | Purpose | +|---------|---------| +| `api` | FastAPI web server (port 8000) | +| `worker` | Celery background task worker | +| `redis` | Message broker for Celery | +| `gotenberg` | PDF conversion (LibreOffice headless) | +| `meilisearch` | Full-text search engine (port 7700) | ### Step 4: Verify the Installation -Access the web interface at `http://localhost:8000` and the API documentation at `http://localhost:8000/docs`. +Access the web interface at `http://localhost:8000` and the API docs at `http://localhost:8000/docs`. + +--- + +## Kubernetes / Helm Deployment + +The Helm chart at `helm/docuelevate/` packages all components into a single, configurable release. It supports: + +- Multiple replicas for the API and Worker +- Horizontal Pod Autoscaling (HPA) +- Bundled or external Redis +- Persistent volumes for workdir and Meilisearch data +- Alembic database migration Job (pre-install/upgrade hook) +- TLS Ingress via any controller (nginx, Traefik, etc.) + +### Prerequisites + +- Kubernetes 1.24+ +- Helm 3.10+ +- A storage class that supports **ReadWriteMany** (e.g. NFS, CephFS, Azure Files, EFS) for the shared workdir PVC when running multiple replicas. Single-replica clusters can use `ReadWriteOnce`. +- A PostgreSQL database (strongly recommended over SQLite for multi-replica). + +### Quick Start + +```bash +# 1. Add the Bitnami chart repository (needed for bundled Redis) +helm repo add bitnami https://charts.bitnami.com/bitnami +helm repo update + +# 2. Update chart dependencies +helm dependency update ./helm/docuelevate + +# 3. Install with a minimal values override +helm install docuelevate ./helm/docuelevate \ + --namespace docuelevate --create-namespace \ + --set secrets.DATABASE_URL="postgresql://user:pass@postgres:5432/docuelevate" \ + --set secrets.SESSION_SECRET="$(openssl rand -hex 32)" \ + --set secrets.OPENAI_API_KEY="sk-..." \ + --set secrets.AZURE_AI_KEY="..." \ + --set env.AZURE_ENDPOINT="https://my-resource.cognitiveservices.azure.com/" \ + --set env.EXTERNAL_HOSTNAME="docuelevate.example.com" +``` + +### Values Reference + +The full list of configurable values is in [`helm/docuelevate/values.yaml`](../helm/docuelevate/values.yaml). Key sections: + +#### Container Image + +```yaml +image: + repository: ghcr.io/christianlouis/docuelevate + tag: "" # defaults to chart appVersion + pullPolicy: IfNotPresent +``` + +#### Non-Secret Config (`env`) + +```yaml +env: + WORKDIR: /workdir + AI_PROVIDER: openai + OPENAI_MODEL: gpt-4o-mini + AZURE_REGION: eastus + AZURE_ENDPOINT: "https://my-resource.cognitiveservices.azure.com/" + MEILISEARCH_URL: http://docuelevate-meilisearch:7700 # auto-resolved from service name + ENABLE_SEARCH: "true" + AUTH_ENABLED: "true" + EXTERNAL_HOSTNAME: docuelevate.example.com +``` + +#### Secrets (`secrets`) + +All secrets are stored in a Kubernetes `Secret` and injected as environment variables. + +```yaml +secrets: + DATABASE_URL: "postgresql://user:pass@postgres:5432/docuelevate" + SESSION_SECRET: "" + OPENAI_API_KEY: "sk-..." + AZURE_AI_KEY: "..." + MEILISEARCH_API_KEY: "" # leave blank for unauthenticated dev Meilisearch + # Storage provider secrets ... +``` + +> **Tip:** In production use an external secret manager (Vault, ESO, Sealed Secrets) and reference the secret by name instead of embedding values in values.yaml. + +#### Replicas & Autoscaling + +```yaml +api: + replicaCount: 2 + autoscaling: + enabled: true + minReplicas: 2 + maxReplicas: 8 + targetCPUUtilizationPercentage: 70 + +worker: + replicaCount: 2 + autoscaling: + enabled: true + minReplicas: 2 + maxReplicas: 10 + targetCPUUtilizationPercentage: 75 +``` + +#### Shared Workdir PVC + +```yaml +workdir: + persistence: + enabled: true + accessMode: ReadWriteMany # RWX required for multi-replica + size: 20Gi + storageClass: "nfs-client" # or leave blank for cluster default +``` + +#### Ingress (nginx example) + +```yaml +ingress: + enabled: true + className: nginx + annotations: + nginx.ingress.kubernetes.io/proxy-body-size: "1g" + cert-manager.io/cluster-issuer: letsencrypt-prod + hosts: + - host: docuelevate.example.com + paths: + - path: / + pathType: Prefix + tls: + - secretName: docuelevate-tls + hosts: + - docuelevate.example.com +``` + +#### External Redis + +```yaml +redis: + enabled: false # disable bundled Redis + +externalRedis: + url: "redis://my-redis-host:6379/0" +``` + +#### Meilisearch + +The bundled Meilisearch deployment is a single-replica, persistent StatefulSet-equivalent. For production, consider [Meilisearch Cloud](https://www.meilisearch.com/cloud) and point `env.MEILISEARCH_URL` at it. + +```yaml +meilisearch: + enabled: true + persistence: + enabled: true + size: 10Gi +``` + +### Upgrading + +```bash +helm upgrade docuelevate ./helm/docuelevate \ + --namespace docuelevate \ + -f my-values.yaml +``` + +The pre-upgrade hook runs `alembic upgrade head` automatically before the new pods start. + +### Uninstalling + +```bash +helm uninstall docuelevate --namespace docuelevate +# PVCs are NOT deleted automatically — remove manually if desired: +kubectl delete pvc -l app.kubernetes.io/instance=docuelevate -n docuelevate +``` + +### Kubernetes Architecture Diagram + +``` +Internet + │ + ▼ +[Ingress / LoadBalancer] + │ + ▼ +[API Deployment] ─────── [Worker Deployment] + │ │ │ │ + │ └── shared PVC ──┘ │ + │ (workdir) │ + ▼ ▼ +[Redis Service] [Gotenberg Service] + │ +[Meilisearch Service] +``` + +--- ## Production Considerations -### Security Headers +### Database -DocuElevate includes built-in support for HTTP security headers to improve browser-side security. **These headers are disabled by default** since most deployments use a reverse proxy (Traefik, Nginx, etc.) that already adds these headers. +SQLite is fine for development but **not recommended for multi-replica production** deployments because it cannot be shared safely across pods. Use PostgreSQL: -#### Supported Security Headers - -- **Strict-Transport-Security (HSTS)**: Forces browsers to use HTTPS for all future requests -- **Content-Security-Policy (CSP)**: Controls which resources browsers are allowed to load -- **X-Frame-Options**: Prevents the page from being loaded in frames (clickjacking protection) -- **X-Content-Type-Options**: Prevents browsers from MIME-sniffing responses - -#### Reverse Proxy Deployment (Traefik, Nginx, etc.) - DEFAULT - -**Most deployments use a reverse proxy**, which is why security headers are disabled by default in DocuElevate. The reverse proxy should add these headers. - -```bash -# In .env file (or omit - this is the default) -SECURITY_HEADERS_ENABLED=false +``` +DATABASE_URL=postgresql://docuelevate:secret@postgres-host:5432/docuelevate ``` -##### Traefik Configuration Example +### Security Headers -Traefik can add security headers using middleware. Create a `docker-compose.yaml` with Traefik labels: +DocuElevate's built-in security headers are **disabled by default** since most deployments use a reverse proxy that already adds them. + +```bash +# Enable only if running without a reverse proxy +SECURITY_HEADERS_ENABLED=true +``` + +#### Traefik (Docker Compose) example ```yaml services: @@ -85,35 +289,25 @@ services: - "traefik.enable=true" - "traefik.http.routers.docuelevate.rule=Host(`docuelevate.example.com`)" - "traefik.http.routers.docuelevate.entrypoints=websecure" - - "traefik.http.routers.docuelevate.tls=true" - "traefik.http.routers.docuelevate.tls.certresolver=letsencrypt" - # Security headers middleware - "traefik.http.routers.docuelevate.middlewares=security-headers@docker" - "traefik.http.middlewares.security-headers.headers.stsSeconds=31536000" - "traefik.http.middlewares.security-headers.headers.stsIncludeSubdomains=true" - - "traefik.http.middlewares.security-headers.headers.contentSecurityPolicy=default-src 'self'; script-src 'self' 'unsafe-inline'; style-src 'self' 'unsafe-inline'; img-src 'self' data: https:; font-src 'self' data:;" - - "traefik.http.middlewares.security-headers.headers.customFrameOptionsValue=DENY" - "traefik.http.middlewares.security-headers.headers.contentTypeNosniff=true" + - "traefik.http.middlewares.security-headers.headers.customFrameOptionsValue=DENY" ``` -Then set `SECURITY_HEADERS_ENABLED=false` in your `.env` file. - -##### Nginx Configuration Example - -Add security headers to your Nginx configuration: +#### Nginx example ```nginx server { listen 443 ssl http2; server_name docuelevate.example.com; - # SSL configuration ssl_certificate /etc/nginx/ssl/cert.pem; ssl_certificate_key /etc/nginx/ssl/key.pem; - # Security headers add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always; - add_header Content-Security-Policy "default-src 'self'; script-src 'self' 'unsafe-inline'; style-src 'self' 'unsafe-inline'; img-src 'self' data: https:; font-src 'self' data:;" always; add_header X-Frame-Options "DENY" always; add_header X-Content-Type-Options "nosniff" always; @@ -123,144 +317,109 @@ server { proxy_set_header X-Real-IP $remote_addr; proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; proxy_set_header X-Forwarded-Proto $scheme; + client_max_body_size 1g; } } ``` -Then keep `SECURITY_HEADERS_ENABLED=false` in your `.env` file (or omit it, as this is the default). - -#### Direct Deployment (No Reverse Proxy) - -If you're running DocuElevate **directly without a reverse proxy**, enable security headers: - -```bash -# In .env file -SECURITY_HEADERS_ENABLED=true -``` - -You can also configure individual headers: - -```bash -SECURITY_HEADER_HSTS_ENABLED=true -SECURITY_HEADER_CSP_ENABLED=true -SECURITY_HEADER_X_FRAME_OPTIONS_ENABLED=true -SECURITY_HEADER_X_CONTENT_TYPE_OPTIONS_ENABLED=true -``` - -**Note**: HSTS only works when serving content over HTTPS. If using HTTP for development, you can disable it: - -```bash -SECURITY_HEADER_HSTS_ENABLED=false -``` - -#### Customizing Security Headers - -If you enable security headers, you can customize individual header values in your `.env` file: - -```bash -# Customize HSTS (e.g., shorter duration for testing) -SECURITY_HEADER_HSTS_VALUE="max-age=300" - -# Customize CSP (e.g., allow specific external domains) -SECURITY_HEADER_CSP_VALUE="default-src 'self'; script-src 'self' https://cdn.example.com; style-src 'self' 'unsafe-inline';" - -# Allow framing from same origin -SECURITY_HEADER_X_FRAME_OPTIONS_VALUE="SAMEORIGIN" -``` - -#### Security Considerations - -1. **HSTS and HTTPS**: HSTS only works over HTTPS. Ensure you have a valid SSL certificate before enabling HSTS. -2. **CSP Testing**: The default CSP policy allows inline scripts and styles for compatibility. Test thoroughly before tightening. -3. **Content-Security-Policy**: The default policy allows `'unsafe-inline'` for scripts and styles for compatibility with Tailwind CSS and inline JavaScript. For stricter security, consider using nonces or hashes. -4. **X-Frame-Options**: Set to `DENY` by default. Change to `SAMEORIGIN` if you need to embed DocuElevate in iframes on the same domain. - -See the [Configuration Guide](ConfigurationGuide.md) for all security header options. - -### Reverse Proxy Setup - -For production use, we recommend setting up a reverse proxy (like Nginx or Traefik) to handle HTTPS and domain routing: - -```nginx -server { - listen 80; - server_name docuelevate.example.com; - - location / { - proxy_pass http://localhost:8000; - proxy_set_header Host $host; - proxy_set_header X-Real-IP $remote_addr; - proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; - proxy_set_header X-Forwarded-Proto $scheme; - } -} -``` - -### Persistent Storage - -The Docker setup uses volumes for persistent storage. For production, consider: +### Storage ```yaml +# Docker Compose volumes: - /path/to/persistent/storage:/workdir + +# Helm — use a RWX storage class for multi-replica +workdir: + persistence: + size: 20Gi + accessMode: ReadWriteMany + storageClass: "nfs-client" ``` -### Security +### General Security Checklist 1. **Always use HTTPS** in production -2. Enable authentication by setting `AUTH_ENABLED=true` -3. Use strong passwords for all services -4. Limit access to the Docker host -5. Regularly update the application and dependencies +2. Set `AUTH_ENABLED=true` and use a strong `SESSION_SECRET` +3. Rotate API keys and secrets regularly — see the [Credential Rotation Guide](CredentialRotationGuide.md) +4. Limit network access to Redis and Meilisearch (both should be internal-only) +5. Regularly update the container image to pick up dependency patches + +--- ## Scaling -For high-volume deployments: +### Docker Compose -1. Increase worker processes by adding more worker containers: +Add more worker containers: ```yaml worker: - image: christianlouis/document-processor:latest deploy: replicas: 3 ``` -2. Consider using dedicated Redis and database servers -3. Monitor system performance and adjust resources as needed +### Kubernetes / Helm + +Enable HPA: + +```yaml +api: + autoscaling: + enabled: true + minReplicas: 2 + maxReplicas: 8 + +worker: + autoscaling: + enabled: true + minReplicas: 2 + maxReplicas: 10 +``` + +--- ## Monitoring -Monitor your DocuElevate deployment using: +- **Docker Compose**: `docker-compose logs -f`, `docker stats` +- **Kubernetes**: `kubectl logs -l app.kubernetes.io/component=api -f` +- **Prometheus / Grafana**: Scrape the `/api/health` endpoint for readiness; add custom metrics as needed. +- **Uptime Kuma**: Set `UPTIME_KUMA_URL` to your push URL for heartbeat monitoring. -- Docker's built-in logging: `docker-compose logs -f` -- Container metrics: `docker stats` -- External monitoring tools like Prometheus and Grafana +--- ## Backup Procedures -Regularly back up the following: +Regularly back up: -1. The `/workdir` directory containing all processed documents -2. The database file (if using SQLite) or database contents (if using another DBMS) -3. The `.env` configuration file +1. The `/workdir` volume (all processed documents and originals) +2. The database (PostgreSQL `pg_dump` or SQLite file) +3. The Meilisearch data directory (`/meili_data`) +4. Your `.env` / Helm values file (store securely, it contains secrets) + +--- ## Updates -To update DocuElevate to a newer version: +### Docker Compose ```bash -# Pull the latest changes git pull - -# Pull the latest Docker images docker-compose pull - -# Restart the services -docker-compose down -docker-compose up -d +docker-compose down && docker-compose up -d ``` +### Helm + +```bash +helm repo update # if using a hosted chart repository +helm upgrade docuelevate ./helm/docuelevate --namespace docuelevate -f my-values.yaml +``` + +The migration Job runs automatically on every `helm upgrade`. + +--- + ## Troubleshooting -See the [Troubleshooting](Troubleshooting.md) guide for common deployment issues and solutions. +See the [Troubleshooting Guide](Troubleshooting.md) for common issues and solutions. diff --git a/frontend/templates/files.html b/frontend/templates/files.html index d277d30e..ab44dbb5 100644 --- a/frontend/templates/files.html +++ b/frontend/templates/files.html @@ -418,6 +418,47 @@ + +
+
+ +
+ + + +
+
+
+ + + +