fix for pdf encoding in account_email_to_pdf

This commit is contained in:
Marc Durepos 2025-03-28 08:00:20 -04:00
parent 640326629a
commit 72530d8c35
2 changed files with 276 additions and 4 deletions

View file

@ -62,6 +62,37 @@ class AccountMove(models.Model):
Returns:
bytes: PDF content as bytes or False if conversion failed
"""
# Check if the content is plain text (not HTML)
# More comprehensive check for HTML content
is_html = bool(re.search(r"<html", html_content, re.IGNORECASE))
_logger.info(f"Input content length: {len(html_content)}")
_logger.info(f"Is content already HTML? {is_html}")
if not is_html:
_logger.info(
"Content appears to be plain text or not a complete HTML document, wrapping in HTML tags"
)
# Escape the content if it's not already HTML
escaped_content = escape(html_content)
_logger.info(f"Escaped content length: {len(escaped_content)}")
# Wrap the content in basic HTML structure with proper styling for plain text
html_content = f"""
<!DOCTYPE html>
<html>
<head>
<meta charset="UTF-8">
<style>
body {{ font-family: Arial, sans-serif; margin: 20px; }}
pre {{ white-space: pre-wrap; font-family: monospace; background-color: #f9f9f9; padding: 10px; }}
</style>
</head>
<body>
<pre>{escaped_content}</pre>
</body>
</html>
"""
_logger.info(f"Wrapped HTML content length: {len(html_content)}")
# Check if wkhtmltopdf is installed
wkhtmltopdf_bin = find_in_path("wkhtmltopdf")
@ -80,7 +111,9 @@ class AccountMove(models.Model):
try:
# Write the HTML content to the temporary file
with closing(os.fdopen(html_file_fd, "wb")) as html_file:
html_file.write(html_content.encode("utf-8"))
encoded_content = html_content.encode("utf-8")
_logger.info(f"Encoded HTML content length: {len(encoded_content)}")
html_file.write(encoded_content)
# Close the PDF file descriptor as wkhtmltopdf will write to it
os.close(pdf_file_fd)
@ -96,15 +129,24 @@ class AccountMove(models.Model):
command.append(html_file_path)
command.append(pdf_file_path)
_logger.info(f"Running wkhtmltopdf command: {' '.join(command)}")
process = subprocess.Popen(
command, stdout=subprocess.PIPE, stderr=subprocess.PIPE
)
_, err = process.communicate()
out, err = process.communicate()
if process.returncode != 0:
_logger.error(
"wkhtmltopdf failed with error code %s: %s", process.returncode, err
)
if out:
_logger.info(
f"wkhtmltopdf stdout: {out.decode('utf-8', errors='replace')[:200]}"
)
if err:
_logger.error(
f"wkhtmltopdf stderr: {err.decode('utf-8', errors='replace')}"
)
return False
# Read the generated PDF
@ -112,6 +154,12 @@ class AccountMove(models.Model):
with open(pdf_file_path, "rb") as pdf_file:
pdf_content = pdf_file.read()
_logger.info(
f"Successfully read PDF file, size: {len(pdf_content)} bytes"
)
if pdf_content:
_logger.info(f"PDF file starts with: {pdf_content[:20]}")
return pdf_content
except Exception as e:
_logger.exception("Error reading generated PDF file: %s", e)
@ -139,6 +187,8 @@ class AccountMove(models.Model):
dict: Dictionary with keys `name` and `datas` containing the name and
base64 encoded data of the attachment
"""
# Log message dict keys for debugging
_logger.info(f"Message dict keys: {list(message_dict.keys())}")
# Extract email details
email_from = message_dict.get("email_from", "Unknown Sender")
email_date = message_dict.get("date", datetime.now())
@ -154,6 +204,58 @@ class AccountMove(models.Model):
_logger.warning("Email body is empty, using placeholder content")
body = "<p>This email did not contain any body content.</p>"
# Check if the body is plain text based on content-type
# The content type can be found in the message headers or directly in the message_dict
content_type = ""
# Try to get content type from various places in the message dict
if "content-type" in message_dict:
content_type = message_dict["content-type"].lower()
_logger.info(f"Found content-type directly in message_dict: {content_type}")
# Try to get from headers
headers = message_dict.get("headers", {})
_logger.info(f"Headers type: {type(headers)}")
if not content_type and headers:
if isinstance(headers, dict):
content_type = headers.get("Content-Type", "").lower()
_logger.info(f"Found Content-Type in headers dict: {content_type}")
elif isinstance(headers, list):
_logger.info(f"Headers is a list with {len(headers)} items")
for header in headers:
_logger.info(f"Header item: {header}")
if (
isinstance(header, tuple)
and len(header) >= 2
and header[0].lower() == "content-type"
):
content_type = header[1].lower()
_logger.info(
f"Found Content-Type in headers list: {content_type}"
)
break
# Try to determine from the body content if we still don't have a content type
if not content_type:
if re.search(r"<html", body, re.IGNORECASE):
content_type = "text/html"
_logger.info("Determined content type as HTML from body content")
else:
content_type = "text/plain"
_logger.info("Defaulting to plain text content type")
_logger.info(f"Final email content type determination: {content_type}")
# Convert plain text to HTML if needed
if "text/plain" in content_type:
_logger.info("Converting plain text email body to HTML")
# Replace newlines with <br> tags and wrap in paragraph tags
# Escape HTML special characters to prevent injection
body = f"<div style='white-space: pre-wrap; font-family: monospace;'>{escape(body)}</div>"
_logger.info(f"Converted plain text body length: {len(body)}")
_logger.info(f"First 100 chars of converted body: {body[:100]}")
# Create HTML content for the PDF
html_content = f"""
<html>
@ -181,16 +283,29 @@ class AccountMove(models.Model):
"""
# Convert HTML to PDF
_logger.info(f"HTML content length before PDF conversion: {len(html_content)}")
_logger.info(f"First 200 chars of HTML content: {html_content[:200]}")
pdf_content = self._html_to_pdf(html_content)
if not pdf_content:
if pdf_content:
_logger.info(f"PDF generation successful, size: {len(pdf_content)} bytes")
_logger.info(f"PDF starts with: {pdf_content[:20]}")
else:
_logger.error("PDF generation failed")
return False
# Create a proper ir.attachment record
filename = f"Email_{subject.replace(' ', '_')[:30]}.pdf"
# Note: No need to base64 encode the PDF content here
# When this attachment is added to message_dict["attachments"], Odoo expects raw binary data
# Odoo will handle the base64 encoding when creating the actual ir.attachment record
_logger.info(f"Raw PDF length: {len(pdf_content)} bytes")
attachment = {
"name": filename,
"datas": base64.b64encode(pdf_content),
"datas": pdf_content,
}
return attachment

View file

@ -1,7 +1,10 @@
from odoo.tests.common import TransactionCase
from odoo.tools.misc import find_in_path
from datetime import datetime
import base64
import logging
_logger = logging.getLogger(__name__)
class TestHtmlToPdf(TransactionCase):
@ -99,6 +102,39 @@ class TestHtmlToPdf(TransactionCase):
)
self.assertTrue(len(pdf_content) > 100, "PDF should have reasonable size")
def test_plain_text_to_pdf_conversion(self):
"""Test the conversion of plain text email to PDF."""
if not self.wkhtmltopdf_available:
self.skipTest("wkhtmltopdf not available")
# Simple plain text content
plain_text = "This is a plain text email.\nIt has no HTML formatting.\nJust plain text content.\n\nRegards,\nTest Sender"
# Log the input content
_logger.info(f"Input plain text content: {plain_text}")
# Call the method to convert plain text to PDF
pdf_content = self.account_move._html_to_pdf(plain_text)
# Log the result
if pdf_content:
_logger.info(f"PDF content generated, size: {len(pdf_content)} bytes")
_logger.info(f"PDF starts with: {pdf_content[:20]}")
else:
_logger.error("Failed to generate PDF from plain text")
# Verify the PDF was created
self.assertTrue(
pdf_content, "PDF content should be generated even for plain text"
)
# Verify it's a valid PDF
is_pdf = pdf_content and pdf_content.startswith(b"%PDF-")
_logger.info(f"Is valid PDF: {is_pdf}")
self.assertTrue(is_pdf, "Content should be a valid PDF")
self.assertTrue(len(pdf_content) > 100, "PDF should have reasonable size")
# Integration test
def test_account_email_to_pdf_full_flow(self):
"""Test that an email without attachments is correctly converted to PDF
@ -191,3 +227,124 @@ Content-Transfer-Encoding: 7bit
self.assertEqual(
invoice.move_type, "in_invoice", "The invoice should be an incoming invoice"
)
def test_plain_text_email_to_pdf_full_flow(self):
"""Test that a plain text email without attachments is correctly converted to PDF
and an account move is created.
This test simulates the full flow of a plain text email being received through
the mail alias and verifies that an account move is created with a PDF
attachment generated from the email content.
"""
_logger.info("Starting test_plain_text_email_to_pdf_full_flow")
# Get the current timestamp
timestamp = datetime.now().strftime("%Y%m%d%H%M%S")
message_id = f"<test123_{timestamp}@example.com>"
subject = f"Plain Text Invoice from Test Supplier {timestamp}"
date = datetime.now().strftime("%a, %d %b %Y %H:%M:%S +0000")
# Plain text body of the email
plain_text_body = """Invoice Test
This is a test invoice from Test Supplier.
Amount: $100.00
Date: 2025-03-27
Thank you for your business.
"""
# Construct the raw email with proper headers
# Make sure the From header is correctly formatted for Odoo to parse
raw_email = f"""Return-Path: <{self.sender_email}>
X-Original-To: {self.alias_email}
Delivered-To: {self.alias_email}
Received: from mail.example.com (mail.example.com [192.168.1.1])
From: {self.sender_email}
To: {self.alias_email}
Subject: {subject}
Date: {date}
Message-ID: {message_id}
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 7bit
{plain_text_body}"""
# Process the email through the mail gateway
# This simulates what happens when fetchmail receives an email
invoice_count_before = self.env["account.move"].search_count(
[("move_type", "=", "in_invoice")]
)
# Process the email through the mail gateway
_logger.info("Processing plain text email through mail gateway")
_logger.info(f"Raw email length: {len(raw_email)}")
_logger.info(f"Raw email first 200 chars: {raw_email[:200]}")
try:
self.env["mail.thread"].with_context(fetchmail_server_id=1).message_process(
model=None,
message=raw_email,
save_original=True,
strip_attachments=False,
)
_logger.info("Email processing completed successfully")
except Exception as e:
_logger.error(f"Error during email processing: {str(e)}")
import traceback
_logger.error(f"Traceback: {traceback.format_exc()}")
raise
# Count account moves after the test
move_count_after = self.env["account.move"].search_count(
[("move_type", "=", "in_invoice")]
)
self.assertEqual(
move_count_after,
invoice_count_before + 1,
"A new account move should be created from plain text email",
)
# Find the newly created invoice
invoice = self.env["account.move"].search(
[("move_type", "=", "in_invoice")], order="id desc", limit=1
)
self.assertTrue(
invoice, "An account move should be created from plain text email"
)
messages = invoice.message_ids
self.assertEqual(len(messages), 1, "The account move should have one message")
attachment = messages.attachment_ids
self.assertTrue(attachment, "The message should have an attachment")
attachment = attachment.filtered(
lambda a: a.mimetype == "application/pdf" or ".pdf" in a.name
)
self.assertTrue(attachment, "The message should have a PDF attachment")
# Verify PDF content if possible
if attachment:
_logger.info(f"Found PDF attachment: {attachment.name}")
pdf_data = base64.b64decode(attachment.datas) if attachment.datas else b""
_logger.info(f"PDF data size: {len(pdf_data)} bytes")
if pdf_data:
_logger.info(f"PDF data starts with: {pdf_data[:20]}")
is_valid_pdf = pdf_data.startswith(b"%PDF-")
_logger.info(f"Is valid PDF: {is_valid_pdf}")
else:
_logger.error("PDF data is empty")
self.assertTrue(
pdf_data.startswith(b"%PDF-"),
f"Attachment should be a valid PDF even when source is plain text, starts with {pdf_data[:20]}",
)
self.assertTrue(
len(pdf_data) > 100, "PDF from plain text should have reasonable size"
)
else:
_logger.error("No PDF attachment found")