email parsing.

This commit is contained in:
Discsearcher
2026-09-02 00:28:43 -04:00
parent 6bc52c6040
commit 821ff8f954
+94 -3
View File
@@ -12,6 +12,8 @@ from werkzeug.security import generate_password_hash, check_password_hash
import markdown import markdown
import logging import logging
import re import re
from email import message_from_string
from email.mime.text import MIMEText
from whoosh.index import create_in, open_dir from whoosh.index import create_in, open_dir
from whoosh.fields import Schema, TEXT, ID from whoosh.fields import Schema, TEXT, ID
from whoosh.qparser import QueryParser from whoosh.qparser import QueryParser
@@ -123,6 +125,94 @@ def is_authenticated(filepath):
authenticated_files = session.get('authenticated_files', {}) authenticated_files = session.get('authenticated_files', {})
return authenticated_files.get(file_key, False) return authenticated_files.get(file_key, False)
def extract_email_text(email_content):
"""
Extract readable text content from an email message.
Handles multipart messages, various encodings (base64, quoted-printable), and prefers text/plain.
Returns formatted email with headers and body.
"""
try:
msg = message_from_string(email_content)
# Extract key headers in order
headers_order = ['From', 'To', 'Cc', 'Bcc', 'Subject', 'Date']
headers = {}
for header in headers_order:
value = msg.get(header)
if value:
# Clean up header values (remove extra whitespace/newlines)
value = ' '.join(value.split())
headers[header] = value
# Extract body content
body_text = None
body_html = None
# Handle multipart messages
if msg.is_multipart():
for part in msg.walk():
content_type = part.get_content_type()
# Skip container parts and non-text content
if content_type.startswith('multipart'):
continue
# Skip if no content
try:
content = part.get_payload(decode=True)
if not content:
continue
except Exception:
continue
try:
text = content.decode('utf-8', errors='replace')
except Exception:
text = str(content)
text = text.strip()
if not text:
continue
# Prefer text/plain, but save html as fallback
if content_type == 'text/plain' and not body_text:
body_text = text
elif content_type == 'text/html' and not body_html:
body_html = text
else:
# Single part message
try:
body_text = msg.get_payload(decode=True).decode('utf-8', errors='replace').strip()
except Exception:
body_text = msg.get_payload()
# Use text if available, otherwise html
body = body_text if body_text else body_html
if not body:
body = "(No message body found)"
# Format the output
formatted_email = []
# Add headers in order
for header in headers_order:
if header in headers:
formatted_email.append(f"{header}: {headers[header]}")
if headers:
formatted_email.append("") # Blank line between headers and body
# Add body
formatted_email.append(body)
return '\n'.join(formatted_email)
except Exception as e:
logger.error(f"Error parsing email: {e}")
# Return original content if parsing fails
return email_content
# Initialize search index # Initialize search index
def create_search_index(): def create_search_index():
"""Create or open the search index.""" """Create or open the search index."""
@@ -439,20 +529,21 @@ def serve_file(filepath):
# Check if content is base64 encoded # Check if content is base64 encoded
# Base64 content is ASCII and contains only base64 characters # Base64 content is ASCII and contains only base64 characters
is_base64 = False
try: try:
if content.strip() and all(c in 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r' for c in content): if content.strip() and all(c in 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r' for c in content):
decoded = base64.b64decode(content, validate=True).decode('utf-8', errors='ignore') decoded = base64.b64decode(content, validate=True).decode('utf-8', errors='ignore')
# If it looks like an email (has headers), use decoded version # If it looks like an email (has headers), use decoded version
if 'From:' in decoded or 'To:' in decoded or 'Subject:' in decoded: if 'From:' in decoded or 'To:' in decoded or 'Subject:' in decoded:
content = decoded content = decoded
is_base64 = True
except Exception: except Exception:
# Not valid base64, use original content # Not valid base64, use original content
pass pass
# Extract and format the email content
formatted_email = extract_email_text(content)
# Return as text/plain for viewing in browser # Return as text/plain for viewing in browser
response = make_response(content) response = make_response(formatted_email)
response.headers['Content-Type'] = 'text/plain; charset=utf-8' response.headers['Content-Type'] = 'text/plain; charset=utf-8'
response.headers['Content-Disposition'] = f'inline; filename="{os.path.basename(file_path)}"' response.headers['Content-Disposition'] = f'inline; filename="{os.path.basename(file_path)}"'
return response return response