#!/usr/bin/env python3 import logging import mimetypes import os import requests from celery import shared_task from app.config import settings from app.tasks.process_document import process_document from app.utils import log_task_progress logger = logging.getLogger(__name__) @shared_task(bind=True) def convert_to_pdf(self, file_path, original_filename=None): """ Converts a file to PDF using Gotenberg's API. Determines the appropriate Gotenberg endpoint based on the file's MIME type. On success, saves the PDF locally and enqueues it for processing. Args: file_path: Path to the file to convert original_filename: Optional original filename (if different from path basename) """ task_id = self.request.id logger.info(f"[{task_id}] Starting PDF conversion: {file_path}") log_task_progress(task_id, "convert_to_pdf", "in_progress", f"Converting file: {os.path.basename(file_path)}") gotenberg_url = getattr(settings, "gotenberg_url", None) if not gotenberg_url: logger.error(f"[{task_id}] Gotenberg URL is not configured in settings.") log_task_progress(task_id, "convert_to_pdf", "failure", "Gotenberg URL not configured") return # Try to guess the MIME type based on file content and extension mime_type, encoding = mimetypes.guess_type(file_path) file_ext = os.path.splitext(file_path)[1].lower() logger.info(f"[{task_id}] Guessed MIME type for '{file_path}' is: {mime_type}, extension: {file_ext}") log_task_progress(task_id, "detect_file_type", "success", f"File type: {mime_type or file_ext}") # Determine which Gotenberg endpoint to use endpoint = None form_data = {} files = {} # Dictionary mapping file extensions to their handlers OFFICE_EXTENSIONS = { ".doc", ".docx", ".docm", ".dot", ".dotx", ".dotm", # Word ".xls", ".xlsx", ".xlsm", ".xlsb", ".xlt", ".xltx", ".xlw", # Excel ".ppt", ".pptx", ".pptm", ".pps", ".ppsx", ".pot", ".potx", # PowerPoint ".odt", ".ods", ".odp", ".odg", ".odf", # OpenOffice/LibreOffice ".rtf", ".txt", ".csv", # Text formats ".pdf", # PDF (already in PDF format but can be processed) } IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png", ".gif", ".bmp", ".tiff", ".tif", ".webp", ".svg"} HTML_EXTENSIONS = {".html", ".htm"} # Use LibreOffice endpoint for office documents and images if ( (mime_type and "office" in mime_type) or (mime_type and "opendocument" in mime_type) or (mime_type and mime_type.startswith("image/")) or file_ext in OFFICE_EXTENSIONS or file_ext in IMAGE_EXTENSIONS ): endpoint = f"{gotenberg_url}/forms/libreoffice/convert" files = {"files": (os.path.basename(file_path), open(file_path, "rb"))} # Add some quality settings for better PDF output form_data = { "landscape": "false", "exportBookmarks": "true", "exportNotes": "false", "losslessImageCompression": "true", # Use lossless compression for images "pdfa": "PDF/A-2b", # Produce PDF/A-2b compatible output } # Use Chromium endpoint for HTML documents elif (mime_type and mime_type == "text/html") or file_ext in HTML_EXTENSIONS: endpoint = f"{gotenberg_url}/forms/chromium/convert/html" # Gotenberg requires the form field to be exactly 'index.html' # The content filename doesn't matter, just the form field key files = {"index.html": ("index.html", open(file_path, "rb"))} # Add options for better HTML to PDF conversion form_data = { "paperWidth": "8.27", # A4 width in inches "paperHeight": "11.7", # A4 height in inches "marginTop": "0.4", "marginBottom": "0.4", "marginLeft": "0.4", "marginRight": "0.4", "printBackground": "true", "preferCssPageSize": "false", "waitDelay": "2s", # Wait for JavaScript to execute } # Use Markdown route for markdown files elif (mime_type and mime_type in ["text/markdown", "text/x-markdown"]) or file_ext in [".md", ".markdown"]: # For Markdown, we need both the markdown file and an HTML wrapper endpoint = f"{gotenberg_url}/forms/chromium/convert/markdown" # Create a simple HTML wrapper for the markdown # IMPORTANT: The filename in the template must match the key used in the files dictionary markdown_filename = os.path.basename(file_path) html_wrapper = f"""