diff --git a/app.py b/app.py index 97996f2..e6c43ba 100644 --- a/app.py +++ b/app.py @@ -12,6 +12,8 @@ from werkzeug.security import generate_password_hash, check_password_hash import markdown import logging import re +from email import message_from_string +from email.mime.text import MIMEText from whoosh.index import create_in, open_dir from whoosh.fields import Schema, TEXT, ID from whoosh.qparser import QueryParser @@ -123,6 +125,94 @@ def is_authenticated(filepath): authenticated_files = session.get('authenticated_files', {}) return authenticated_files.get(file_key, False) +def extract_email_text(email_content): + """ + Extract readable text content from an email message. + Handles multipart messages, various encodings (base64, quoted-printable), and prefers text/plain. + Returns formatted email with headers and body. + """ + try: + msg = message_from_string(email_content) + + # Extract key headers in order + headers_order = ['From', 'To', 'Cc', 'Bcc', 'Subject', 'Date'] + headers = {} + for header in headers_order: + value = msg.get(header) + if value: + # Clean up header values (remove extra whitespace/newlines) + value = ' '.join(value.split()) + headers[header] = value + + # Extract body content + body_text = None + body_html = None + + # Handle multipart messages + if msg.is_multipart(): + for part in msg.walk(): + content_type = part.get_content_type() + + # Skip container parts and non-text content + if content_type.startswith('multipart'): + continue + + # Skip if no content + try: + content = part.get_payload(decode=True) + if not content: + continue + except Exception: + continue + + try: + text = content.decode('utf-8', errors='replace') + except Exception: + text = str(content) + + text = text.strip() + if not text: + continue + + # Prefer text/plain, but save html as fallback + if content_type == 'text/plain' and not body_text: + body_text = text + elif content_type == 'text/html' and not body_html: + body_html = text + else: + # Single part message + try: + body_text = msg.get_payload(decode=True).decode('utf-8', errors='replace').strip() + except Exception: + body_text = msg.get_payload() + + # Use text if available, otherwise html + body = body_text if body_text else body_html + + if not body: + body = "(No message body found)" + + # Format the output + formatted_email = [] + + # Add headers in order + for header in headers_order: + if header in headers: + formatted_email.append(f"{header}: {headers[header]}") + + if headers: + formatted_email.append("") # Blank line between headers and body + + # Add body + formatted_email.append(body) + + return '\n'.join(formatted_email) + + except Exception as e: + logger.error(f"Error parsing email: {e}") + # Return original content if parsing fails + return email_content + # Initialize search index def create_search_index(): """Create or open the search index.""" @@ -439,20 +529,21 @@ def serve_file(filepath): # Check if content is base64 encoded # Base64 content is ASCII and contains only base64 characters - is_base64 = False try: if content.strip() and all(c in 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r' for c in content): decoded = base64.b64decode(content, validate=True).decode('utf-8', errors='ignore') # If it looks like an email (has headers), use decoded version if 'From:' in decoded or 'To:' in decoded or 'Subject:' in decoded: content = decoded - is_base64 = True except Exception: # Not valid base64, use original content pass + # Extract and format the email content + formatted_email = extract_email_text(content) + # Return as text/plain for viewing in browser - response = make_response(content) + response = make_response(formatted_email) response.headers['Content-Type'] = 'text/plain; charset=utf-8' response.headers['Content-Disposition'] = f'inline; filename="{os.path.basename(file_path)}"' return response