SHA256
email parsing.
This commit is contained in:
@@ -12,6 +12,8 @@ from werkzeug.security import generate_password_hash, check_password_hash
|
||||
import markdown
|
||||
import logging
|
||||
import re
|
||||
from email import message_from_string
|
||||
from email.mime.text import MIMEText
|
||||
from whoosh.index import create_in, open_dir
|
||||
from whoosh.fields import Schema, TEXT, ID
|
||||
from whoosh.qparser import QueryParser
|
||||
@@ -123,6 +125,94 @@ def is_authenticated(filepath):
|
||||
authenticated_files = session.get('authenticated_files', {})
|
||||
return authenticated_files.get(file_key, False)
|
||||
|
||||
def extract_email_text(email_content):
|
||||
"""
|
||||
Extract readable text content from an email message.
|
||||
Handles multipart messages, various encodings (base64, quoted-printable), and prefers text/plain.
|
||||
Returns formatted email with headers and body.
|
||||
"""
|
||||
try:
|
||||
msg = message_from_string(email_content)
|
||||
|
||||
# Extract key headers in order
|
||||
headers_order = ['From', 'To', 'Cc', 'Bcc', 'Subject', 'Date']
|
||||
headers = {}
|
||||
for header in headers_order:
|
||||
value = msg.get(header)
|
||||
if value:
|
||||
# Clean up header values (remove extra whitespace/newlines)
|
||||
value = ' '.join(value.split())
|
||||
headers[header] = value
|
||||
|
||||
# Extract body content
|
||||
body_text = None
|
||||
body_html = None
|
||||
|
||||
# Handle multipart messages
|
||||
if msg.is_multipart():
|
||||
for part in msg.walk():
|
||||
content_type = part.get_content_type()
|
||||
|
||||
# Skip container parts and non-text content
|
||||
if content_type.startswith('multipart'):
|
||||
continue
|
||||
|
||||
# Skip if no content
|
||||
try:
|
||||
content = part.get_payload(decode=True)
|
||||
if not content:
|
||||
continue
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
try:
|
||||
text = content.decode('utf-8', errors='replace')
|
||||
except Exception:
|
||||
text = str(content)
|
||||
|
||||
text = text.strip()
|
||||
if not text:
|
||||
continue
|
||||
|
||||
# Prefer text/plain, but save html as fallback
|
||||
if content_type == 'text/plain' and not body_text:
|
||||
body_text = text
|
||||
elif content_type == 'text/html' and not body_html:
|
||||
body_html = text
|
||||
else:
|
||||
# Single part message
|
||||
try:
|
||||
body_text = msg.get_payload(decode=True).decode('utf-8', errors='replace').strip()
|
||||
except Exception:
|
||||
body_text = msg.get_payload()
|
||||
|
||||
# Use text if available, otherwise html
|
||||
body = body_text if body_text else body_html
|
||||
|
||||
if not body:
|
||||
body = "(No message body found)"
|
||||
|
||||
# Format the output
|
||||
formatted_email = []
|
||||
|
||||
# Add headers in order
|
||||
for header in headers_order:
|
||||
if header in headers:
|
||||
formatted_email.append(f"{header}: {headers[header]}")
|
||||
|
||||
if headers:
|
||||
formatted_email.append("") # Blank line between headers and body
|
||||
|
||||
# Add body
|
||||
formatted_email.append(body)
|
||||
|
||||
return '\n'.join(formatted_email)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing email: {e}")
|
||||
# Return original content if parsing fails
|
||||
return email_content
|
||||
|
||||
# Initialize search index
|
||||
def create_search_index():
|
||||
"""Create or open the search index."""
|
||||
@@ -439,20 +529,21 @@ def serve_file(filepath):
|
||||
|
||||
# Check if content is base64 encoded
|
||||
# Base64 content is ASCII and contains only base64 characters
|
||||
is_base64 = False
|
||||
try:
|
||||
if content.strip() and all(c in 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r' for c in content):
|
||||
decoded = base64.b64decode(content, validate=True).decode('utf-8', errors='ignore')
|
||||
# If it looks like an email (has headers), use decoded version
|
||||
if 'From:' in decoded or 'To:' in decoded or 'Subject:' in decoded:
|
||||
content = decoded
|
||||
is_base64 = True
|
||||
except Exception:
|
||||
# Not valid base64, use original content
|
||||
pass
|
||||
|
||||
# Extract and format the email content
|
||||
formatted_email = extract_email_text(content)
|
||||
|
||||
# Return as text/plain for viewing in browser
|
||||
response = make_response(content)
|
||||
response = make_response(formatted_email)
|
||||
response.headers['Content-Type'] = 'text/plain; charset=utf-8'
|
||||
response.headers['Content-Disposition'] = f'inline; filename="{os.path.basename(file_path)}"'
|
||||
return response
|
||||
|
||||
Reference in New Issue
Block a user