diff --git a/.env.demo b/.env.demo index a24f4427..144b4ee4 100644 --- a/.env.demo +++ b/.env.demo @@ -44,4 +44,11 @@ IMAP2_SSL=true IMAP2_POLL_INTERVAL_MINUTES=10 IMAP2_DELETE_AFTER_PROCESS=false -GOTENBERG_URL=http://gotenberg:3000 \ No newline at end of file +GOTENBERG_URL=http://gotenberg:3000 + +# ** needed for Authentik ** +AUTH_ENABLED=true +SESSION_SECRET= +AUTHENTIK_CLIENT_ID= +AUTHENTIK_CLIENT_SECRET= +AUTHENTIK_CONFIG_URL= \ No newline at end of file diff --git a/app/api.py b/app/api.py new file mode 100644 index 00000000..f768e26a --- /dev/null +++ b/app/api.py @@ -0,0 +1,32 @@ +from fastapi import APIRouter, Request, HTTPException, status +from hashlib import md5 + +router = APIRouter() + +@router.get("/whoami") +async def whoami(request: Request): + """ + Returns user info if logged in, else 401. + Example response: + { + "email": "someone@example.com", + "picture": "https://www.gravatar.com/avatar/..." + } + """ + user = request.session.get("user") + if not user: + raise HTTPException(status_code=401, detail="Not logged in") + + email = user.get("email") + if not email: + raise HTTPException(status_code=400, detail="User has no email in session") + + # Generate Gravatar URL from email + # For more options, see: https://en.gravatar.com/site/implement/images/ + email_hash = md5(email.strip().lower().encode()).hexdigest() + gravatar_url = f"https://www.gravatar.com/avatar/{email_hash}?d=identicon" + + return { + "email": email, + "picture": gravatar_url + } \ No newline at end of file diff --git a/app/auth.py b/app/auth.py new file mode 100644 index 00000000..db3df500 --- /dev/null +++ b/app/auth.py @@ -0,0 +1,71 @@ +# app/auth.py +# app/auth.py +import os +from functools import wraps + +from authlib.integrations.starlette_client import OAuth +from starlette.config import Config +from fastapi import APIRouter, Request, status +from starlette.responses import RedirectResponse + +config = Config(".env") +oauth = OAuth(config) + +AUTH_ENABLED = config("AUTH_ENABLED", cast=bool, default=True) + +if AUTH_ENABLED: + oauth.register( + name="authentik", + client_id=config("AUTHENTIK_CLIENT_ID"), + client_secret=config("AUTHENTIK_CLIENT_SECRET"), + server_metadata_url=config("AUTHENTIK_CONFIG_URL"), + client_kwargs={"scope": "openid profile email"}, + ) + +router = APIRouter() + + +def get_current_user(request: Request): + return request.session.get("user") + + +def require_login(func): + if not AUTH_ENABLED: + return func # no-op + + @wraps(func) + async def wrapper(request: Request, *args, **kwargs): + if not request.session.get("user"): + request.session["redirect_after_login"] = str(request.url) + return RedirectResponse(url="/login", status_code=status.HTTP_302_FOUND) + return await func(request, *args, **kwargs) + + return wrapper + + +if AUTH_ENABLED: + @router.get("/login") + async def login(request: Request): + redirect_uri = request.url_for("auth") + return await oauth.authentik.authorize_redirect(request, redirect_uri) + + @router.get("/auth") + async def auth(request: Request): + token = await oauth.authentik.authorize_access_token(request) + userinfo = token.get("userinfo") + request.session["user"] = dict(userinfo) + redirect_url = request.session.pop("redirect_after_login", "/upload") + return RedirectResponse(url=redirect_url) + + @router.get("/logout") + async def logout(request: Request): + request.session.pop("user", None) + return RedirectResponse(url="/") + + +@router.get("/private") +@require_login +async def private_page(request: Request): + """A protected endpoint that requires login.""" + user = request.session.get("user") # e.g. {"email": "...", ...} + return {"message": f"This is a protected page. Hello {user['email']}!"} diff --git a/app/frontend.py b/app/frontend.py index 9cec1c42..9cc25798 100644 --- a/app/frontend.py +++ b/app/frontend.py @@ -1,7 +1,8 @@ # app/frontend.py (new file or inline in main.py) -from fastapi import APIRouter +from fastapi import APIRouter, Request, status from fastapi.responses import FileResponse from fastapi.staticfiles import StaticFiles +from app.auth import require_login import os router = APIRouter() @@ -14,11 +15,25 @@ frontend_folder = os.path.join(os.path.dirname(__file__), "..", "frontend") router.mount("/static", StaticFiles(directory=frontend_folder), name="static") # 2) For the root route ("/"), return the index.html -@router.get("/ui", response_class=FileResponse) -def serve_ui(): - return os.path.join(frontend_folder, "index.html") + +@router.get("/upload", response_class=FileResponse) +@require_login +async def serve_upload(request: Request): + return os.path.join(frontend_folder, "upload.html") # 3) Serve favicon.ico from the frontend folder @router.get("/favicon.ico", response_class=FileResponse) def favicon(): - return os.path.join(frontend_folder, "favicon.ico") \ No newline at end of file + return os.path.join(frontend_folder, "favicon.ico") + +""" @router.exception_handler(404) +async def custom_404_handler(request: Request, exc): + return FileResponse("frontend/404.html", status_code=status.HTTP_404_NOT_FOUND) """ + +@router.get("/", response_class=FileResponse) +async def serve_index(request: Request): + return os.path.join(frontend_folder, "index.html") + +@router.get("/about", response_class=FileResponse) +async def serve_about(request: Request): + return os.path.join(frontend_folder, "about.html") \ No newline at end of file diff --git a/app/main.py b/app/main.py index aeaa70cc..d255c451 100644 --- a/app/main.py +++ b/app/main.py @@ -1,7 +1,12 @@ #!/usr/bin/env python3 - import os -from fastapi import FastAPI, HTTPException, UploadFile, File + +from fastapi import FastAPI, HTTPException, UploadFile, File, status, Request +from fastapi.responses import FileResponse +from starlette.middleware.sessions import SessionMiddleware +from starlette.config import Config +from starlette.middleware.trustedhost import TrustedHostMiddleware +from uvicorn.middleware.proxy_headers import ProxyHeadersMiddleware from app.database import init_db from app.config import settings from app.tasks.upload_to_s3 import upload_to_s3 @@ -9,17 +14,42 @@ from app.tasks.upload_to_dropbox import upload_to_dropbox from app.tasks.upload_to_paperless import upload_to_paperless from app.tasks.upload_to_nextcloud import upload_to_nextcloud from app.tasks.send_to_all import send_to_all_destinations +from pathlib import Path + +from app.api import router as api_router from app.frontend import router as frontend_router +from app.auth import router as auth_router + + +# Load configuration from .env for the session key +config = Config(".env") +SESSION_SECRET = config( + "SESSION_SECRET", + default="YOUR_DEFAULT_SESSION_SECRET_MUST_BE_32_CHARS_OR_MORE" +) app = FastAPI(title="Document Processing API") + +# 1) Session Middleware (for request.session to work) +app.add_middleware(SessionMiddleware, secret_key=SESSION_SECRET) + +# 2) Respect the X-Forwarded-* headers from Traefik +# so your request.url_for(...) uses https://docparse.hosterra.net +app.add_middleware(ProxyHeadersMiddleware, trusted_hosts="*") + +# 3) (Optional but recommended) Restrict valid hosts: +app.add_middleware(TrustedHostMiddleware, allowed_hosts=[ + "docparse.hosterra.net", + "localhost", + "127.0.0.1" +]) + @app.on_event("startup") def on_startup(): init_db() # Create tables if they don't exist -@app.get("/") -def root(): - return {"message": "Document Processing API"} + @app.post("/process/") def process(file_path: str): @@ -27,13 +57,13 @@ def process(file_path: str): API Endpoint to start document processing. This enqueues the first task (upload_to_s3), which handles the full pipeline. """ - - # If file_path is not absolute, treat it as relative to settings.workdir. if not os.path.isabs(file_path): file_path = os.path.join(settings.workdir, file_path) if not os.path.exists(file_path): - raise HTTPException(status_code=400, detail=f"File {file_path} not found.") + raise HTTPException( + status_code=400, detail=f"File {file_path} not found." + ) task = upload_to_s3.delay(file_path) return {"task_id": task.id, "status": "queued"} @@ -43,7 +73,9 @@ def send_to_dropbox(file_path: str): if not os.path.isabs(file_path): file_path = os.path.join(settings.workdir, 'processed', file_path) if not os.path.exists(file_path): - raise HTTPException(status_code=400, detail=f"File {file_path} not found.") + raise HTTPException( + status_code=400, detail=f"File {file_path} not found." + ) task = upload_to_dropbox.delay(file_path) return {"task_id": task.id, "status": "queued"} @@ -52,7 +84,9 @@ def send_to_paperless(file_path: str): if not os.path.isabs(file_path): file_path = os.path.join(settings.workdir, 'processed', file_path) if not os.path.exists(file_path): - raise HTTPException(status_code=400, detail=f"File {file_path} not found.") + raise HTTPException( + status_code=400, detail=f"File {file_path} not found." + ) task = upload_to_paperless.delay(file_path) return {"task_id": task.id, "status": "queued"} @@ -61,40 +95,38 @@ def send_to_nextcloud(file_path: str): if not os.path.isabs(file_path): file_path = os.path.join(settings.workdir, 'processed', file_path) if not os.path.exists(file_path): - raise HTTPException(status_code=400, detail=f"File {file_path} not found.") + raise HTTPException( + status_code=400, detail=f"File {file_path} not found." + ) task = upload_to_nextcloud.delay(file_path) return {"task_id": task.id, "status": "queued"} - @app.post("/send_to_all_destinations/") def send_to_all_destinations_endpoint(file_path: str): """ Call the aggregator task that sends this file to dropbox, nextcloud, and paperless. """ if not os.path.isabs(file_path): - # If not absolute, assume it's in processed subdir file_path = os.path.join(settings.workdir, 'processed', file_path) if not os.path.exists(file_path): raise HTTPException( - status_code=400, - detail=f"File {file_path} not found." + status_code=400, detail=f"File {file_path} not found." ) task = send_to_all_destinations.delay(file_path) return {"task_id": task.id, "status": "queued", "file_path": file_path} - - @app.post("/processall") def process_all_pdfs_in_workdir(): """ - Finds all .pdf files in /processed - and enqueues them for upload_to_s3. + Finds all .pdf files in and enqueues them for upload_to_s3. """ target_dir = settings.workdir if not os.path.exists(target_dir): - raise HTTPException(status_code=400, detail=f"Directory {target_dir} does not exist.") + raise HTTPException( + status_code=400, detail=f"Directory {target_dir} does not exist." + ) pdf_files = [] for filename in os.listdir(target_dir): @@ -102,13 +134,11 @@ def process_all_pdfs_in_workdir(): pdf_files.append(filename) if not pdf_files: - return {"message": "No PDF files found in processed directory."} + return {"message": "No PDF files found in that directory."} task_ids = [] for pdf in pdf_files: file_path = os.path.join(target_dir, pdf) - # Enqueue upload_to_s3 - from app.tasks.upload_to_s3 import upload_to_s3 task = upload_to_s3.delay(file_path) task_ids.append(task.id) @@ -118,21 +148,33 @@ def process_all_pdfs_in_workdir(): "task_ids": task_ids } -app.include_router(frontend_router) - @app.post("/ui-upload") async def ui_upload(file: UploadFile = File(...)): - # You can store this file in your 'workdir' (like how /process does) or a tmp dir + """Endpoint to accept a user-uploaded file and enqueue it to S3.""" workdir = "/workdir" target_path = os.path.join(workdir, file.filename) - try: with open(target_path, "wb") as f: content = await file.read() f.write(content) except Exception as e: - raise HTTPException(status_code=500, detail=f"Failed to save file: {e}") + raise HTTPException( + status_code=500, + detail=f"Failed to save file: {e}" + ) - # Now you can call your existing Celery flow: task = upload_to_s3.delay(target_path) - return {"task_id": task.id, "status": "queued"} \ No newline at end of file + return {"task_id": task.id, "status": "queued"} + +@app.exception_handler(404) +async def custom_404_handler(request: Request, exc: HTTPException): + return FileResponse( + "/app/frontend/404.html", + status_code=status.HTTP_404_NOT_FOUND + ) + +# Include the frontend and auth routers +app.include_router(frontend_router) +app.include_router(auth_router) +app.include_router(api_router, prefix="/api") + diff --git a/docker-compose.yaml b/docker-compose.yaml index 4e67945d..38295749 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -8,7 +8,7 @@ services: working_dir: /workdir # We'll run uvicorn from the container's /app code - command: ["sh", "-c", "cd /app && uvicorn app.main:app --host 0.0.0.0 --port 8000"] + command: ["sh", "-c", "cd /app && uvicorn app.main:app --host 0.0.0.0 --port 8000 --proxy-headers"] # Environment variables environment: diff --git a/frontend/404.html b/frontend/404.html new file mode 100644 index 00000000..c085430d --- /dev/null +++ b/frontend/404.html @@ -0,0 +1,93 @@ + + + + + 404 - Oops, DocuNova Lost the Page + + + + + + + + + +
+
+

404

+

Oops, we couldn’t find that page!

+

+ It seems DocuNova has misplaced the document you were looking for. Whether it got lost in the cloud or hidden between the files, don’t worry – we’ve got your back. +

+ + ← Return Home + +
+
+ + +
+
+ © 2025 DocuNova. All rights reserved. +
+
+ + + + + diff --git a/frontend/about.html b/frontend/about.html new file mode 100644 index 00000000..718b8513 --- /dev/null +++ b/frontend/about.html @@ -0,0 +1,132 @@ + + + + + About DocuNova + + + + + + + + + +
+

About DocuNova

+

+ Welcome to DocuNova – your modern, intelligent solution for document processing! We’ve built DocuNova to completely transform the way you handle your documents – from upload to extraction, from processing to storage. With cutting-edge technologies and a user-first design, DocuNova makes managing your documents as easy as a click. +

+ + +
+

Our Story

+

+ DocuNova was created with one goal in mind: to simplify and streamline document management for everyone, whether you’re a small startup or a large enterprise. Tired of clunky, outdated systems, we set out to design a platform that is intuitive, flexible, and packed with powerful features. +

+

+ We harness the power of OpenAI for metadata extraction and text refinement, integrate seamlessly with Dropbox, Nextcloud, and Paperless NGX for storage and indexing, leverage Azure Document Intelligence for OCR, and even use Gotenberg for file-to-PDF conversions. And while we’ve implemented AWS S3 for now, we’re always evolving! +

+
+ + +
+

Key Features

+
    +
  • Simple and secure file uploads with drag & drop support
  • +
  • Automated metadata extraction, indexing, and version control
  • +
  • Integration with popular cloud services and storage platforms
  • +
  • OCR and intelligent document processing powered by AI
  • +
  • IMAP integration for automated document fetching
  • +
  • Highly configurable via environment variables for custom workflows
  • +
  • Docker-ready for easy deployment and scalability
  • +
+
+ + +
+

Meet the Creator

+

+ DocuNova is passionately developed by Christian Krakau-Louis, a visionary committed to solving real-world challenges with innovative technology. His dedication and expertise ensure that every feature in DocuNova is crafted with you in mind. +

+
+ + +
+

Get Involved

+

+ Want to dive into the code, contribute ideas, or simply check out the magic behind DocuNova? Visit our GitHub repository to see the project in action! +

+ + GitHub Logo + View DocuNova on GitHub + +
+
+ + +
+
+ © 2025 DocuNova. All rights reserved. +
+
+ + + + + diff --git a/frontend/index.html b/frontend/index.html index b22c3d7d..6c183692 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -1,89 +1,108 @@ - - - - - - Document Processor - Upload - - - - - -

Upload a File

- -
-

- Drag & drop a file here, or click to select a file. -

- -
- -
- - - - + + + + + DocuNova - Intelligent Document Processing + + + + + + + + + +
+

Welcome to DocuNova

+

+ Your intelligent solution for processing, managing, and organizing documents effortlessly. +

+ +
+
+

Upload Documents

+

+ Quickly upload and process your files with our user-friendly interface. +

+ + Get Started → + +
+
+

Manage Your Files

+

+ View, organize, and collaborate on your processed files in one central hub. +

+ + View Files → + +
+
+
+ + +
+
+ © 2025 DocuNova. All rights reserved. +
+
+ + + + + diff --git a/frontend/upload.html b/frontend/upload.html new file mode 100644 index 00000000..eec9ca34 --- /dev/null +++ b/frontend/upload.html @@ -0,0 +1,157 @@ + + + + + DocuNova - Upload + + + + + + + + + +
+

Upload a File

+ +
+

+ Drag & drop a file here, or click to select a file. +

+ +
+ +
+
+ + +
+
+ © 2025 DocuNova. All rights reserved. +
+
+ + + + + diff --git a/requirements.txt b/requirements.txt index e61f348f..82bcc5fd 100644 --- a/requirements.txt +++ b/requirements.txt @@ -12,4 +12,7 @@ openai pymupdf requests dropbox -azure-ai-documentintelligence \ No newline at end of file +azure-ai-documentintelligence +authlib +python-dotenv +starlette \ No newline at end of file