#!/usr/bin/env python3 import os import requests import logging import mimetypes import json from celery import shared_task from app.config import settings from app.tasks.process_document import process_document logger = logging.getLogger(__name__) @shared_task def convert_to_pdf(file_path): """ Converts a file to PDF using Gotenberg's API. Determines the appropriate Gotenberg endpoint based on the file's MIME type. On success, saves the PDF locally and enqueues it for processing. """ gotenberg_url = getattr(settings, "gotenberg_url", None) if not gotenberg_url: logger.error("Gotenberg URL is not configured in settings.") return # Try to guess the MIME type based on file content and extension mime_type, encoding = mimetypes.guess_type(file_path) file_ext = os.path.splitext(file_path)[1].lower() logger.info(f"Guessed MIME type for '{file_path}' is: {mime_type}, extension: {file_ext}") # Determine which Gotenberg endpoint to use endpoint = None form_data = {} files = {} # Dictionary mapping file extensions to their handlers OFFICE_EXTENSIONS = { '.doc', '.docx', '.docm', '.dot', '.dotx', '.dotm', # Word '.xls', '.xlsx', '.xlsm', '.xlsb', '.xlt', '.xltx', '.xlw', # Excel '.ppt', '.pptx', '.pptm', '.pps', '.ppsx', '.pot', '.potx', # PowerPoint '.odt', '.ods', '.odp', '.odg', '.odf', # OpenOffice/LibreOffice '.rtf', '.txt', '.csv', # Text formats '.pdf', # PDF (already in PDF format but can be processed) } IMAGE_EXTENSIONS = { '.jpg', '.jpeg', '.png', '.gif', '.bmp', '.tiff', '.tif', '.webp', '.svg' } HTML_EXTENSIONS = { '.html', '.htm' } # Use LibreOffice endpoint for office documents and images if (mime_type and 'office' in mime_type) or \ (mime_type and 'opendocument' in mime_type) or \ (mime_type and mime_type.startswith('image/')) or \ file_ext in OFFICE_EXTENSIONS or \ file_ext in IMAGE_EXTENSIONS: endpoint = f"{gotenberg_url}/forms/libreoffice/convert" files = {'files': (os.path.basename(file_path), open(file_path, 'rb'))} # Add some quality settings for better PDF output form_data = { 'landscape': 'false', 'exportBookmarks': 'true', 'exportNotes': 'false', 'losslessImageCompression': 'true', # Use lossless compression for images 'pdfa': 'PDF/A-2b', # Produce PDF/A-2b compatible output } # Use Chromium endpoint for HTML documents elif (mime_type and mime_type == 'text/html') or file_ext in HTML_EXTENSIONS: endpoint = f"{gotenberg_url}/forms/chromium/convert/html" # Gotenberg requires the form field to be exactly 'index.html' # The content filename doesn't matter, just the form field key files = {'index.html': ('index.html', open(file_path, 'rb'))} # Add options for better HTML to PDF conversion form_data = { 'paperWidth': '8.27', # A4 width in inches 'paperHeight': '11.7', # A4 height in inches 'marginTop': '0.4', 'marginBottom': '0.4', 'marginLeft': '0.4', 'marginRight': '0.4', 'printBackground': 'true', 'preferCssPageSize': 'false', 'waitDelay': '2s', # Wait for JavaScript to execute } # Use Markdown route for markdown files elif (mime_type and mime_type in ['text/markdown', 'text/x-markdown']) or file_ext in ['.md', '.markdown']: # For Markdown, we need both the markdown file and an HTML wrapper endpoint = f"{gotenberg_url}/forms/chromium/convert/markdown" # Create a simple HTML wrapper for the markdown # IMPORTANT: The filename in the template must match the key used in the files dictionary markdown_filename = os.path.basename(file_path) html_wrapper = f"""