Files
markmywords/app.py
T
2026-09-02 00:39:21 -04:00

822 lines
30 KiB
Python

#!/usr/bin/env python3
"""
markmywords - Self-hosted markdown viewer with search and encryption
"""
import os
import json
from pathlib import Path
from datetime import datetime
from flask import Flask, render_template, request, jsonify, send_file, session, redirect, url_for, make_response
from werkzeug.security import generate_password_hash, check_password_hash
import markdown
import logging
import re
from email import message_from_string
from email.mime.text import MIMEText
from whoosh.index import create_in, open_dir
from whoosh.fields import Schema, TEXT, ID
from whoosh.qparser import QueryParser
from cryptography.fernet import Fernet
from cryptography.hazmat.primitives import hashes
from cryptography.hazmat.primitives.kdf.pbkdf2 import PBKDF2HMAC
from cryptography.hazmat.backends import default_backend
import base64
from apscheduler.schedulers.background import BackgroundScheduler
from apscheduler.triggers.interval import IntervalTrigger
app = Flask(__name__)
app.secret_key = os.environ.get('FLASK_SECRET_KEY', 'dev-secret-key-change-in-production')
# Configuration
CONTENT_DIR = os.environ.get('CONTENT_DIR', "/content")
INDEX_DIR = "/data/search_index"
PASSWORD_FILE = "/config/page_passwords.json"
MARKDOWN_EXTENSIONS = ['.md', '.markdown']
ENCRYPTION_ENABLED = os.environ.get('ENCRYPTION_ENABLED', 'true').lower() == 'true'
# Setup logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# Ensure directories exist
os.makedirs(CONTENT_DIR, exist_ok=True)
os.makedirs(INDEX_DIR, exist_ok=True)
os.makedirs(os.path.dirname(PASSWORD_FILE), exist_ok=True)
def load_passwords():
"""Load password database from file."""
if not os.path.exists(PASSWORD_FILE):
return {}
try:
with open(PASSWORD_FILE, 'r') as f:
return json.load(f)
except Exception as e:
logger.error(f"Error loading passwords: {e}")
return {}
def save_passwords(passwords):
"""Save password database to file."""
try:
os.makedirs(os.path.dirname(PASSWORD_FILE), exist_ok=True)
with open(PASSWORD_FILE, 'w') as f:
json.dump(passwords, f, indent=2)
os.chmod(PASSWORD_FILE, 0o600) # Restrict file permissions
return True
except Exception as e:
logger.error(f"Error saving passwords: {e}")
return False
def derive_encryption_key(password):
"""Derive an encryption key from a password."""
salt = b'markmywords_salt' # Fixed salt for consistency
kdf = PBKDF2HMAC(
algorithm=hashes.SHA256(),
length=32,
salt=salt,
iterations=100000,
backend=default_backend()
)
key = base64.urlsafe_b64encode(kdf.derive(password.encode()))
return key
def encrypt_file_content(content, password):
"""Encrypt file content using password-derived key."""
try:
key = derive_encryption_key(password)
f = Fernet(key)
encrypted = f.encrypt(content.encode('utf-8'))
return encrypted.decode('utf-8')
except Exception as e:
logger.error(f"Error encrypting content: {e}")
return None
def decrypt_file_content(encrypted_content, password):
"""Decrypt file content using password-derived key."""
try:
key = derive_encryption_key(password)
f = Fernet(key)
decrypted = f.decrypt(encrypted_content.encode('utf-8'))
return decrypted.decode('utf-8')
except Exception as e:
logger.error(f"Error decrypting content: {e}")
return None
def is_encrypted_file(filepath):
"""Check if a file is encrypted (has .enc extension)."""
return filepath.endswith('.enc')
def get_file_key(filepath):
"""Generate a consistent key for a file."""
return filepath
def is_file_protected(filepath):
"""Check if a file is password protected."""
passwords = load_passwords()
file_key = get_file_key(filepath)
return passwords.get(file_key, {}).get('protected', False)
def is_authenticated(filepath):
"""Check if user is authenticated for a protected file."""
if not is_file_protected(filepath):
return True # Not protected, so allowed
file_key = get_file_key(filepath)
authenticated_files = session.get('authenticated_files', {})
return authenticated_files.get(file_key, False)
def extract_email_text(email_content):
"""
Extract readable text content from an email message.
Handles multipart messages, various encodings (base64, quoted-printable), and prefers text/plain.
Returns formatted email with headers and body.
"""
try:
msg = message_from_string(email_content)
# Extract key headers in order
headers_order = ['From', 'To', 'Cc', 'Bcc', 'Subject', 'Date']
headers = {}
for header in headers_order:
value = msg.get(header)
if value:
# Clean up header values (remove extra whitespace/newlines)
value = ' '.join(value.split())
headers[header] = value
# Extract body content
body_text = None
body_html = None
# Handle multipart messages
if msg.is_multipart():
for part in msg.walk():
content_type = part.get_content_type()
# Skip container parts and non-text content
if content_type.startswith('multipart'):
continue
# Skip if no content
try:
content = part.get_payload(decode=True)
if not content:
continue
except Exception:
continue
try:
text = content.decode('utf-8', errors='replace')
except Exception:
text = str(content)
text = text.strip()
if not text:
continue
# Prefer text/plain, but save html as fallback
if content_type == 'text/plain' and not body_text:
body_text = text
elif content_type == 'text/html' and not body_html:
body_html = text
else:
# Single part message
try:
body_text = msg.get_payload(decode=True).decode('utf-8', errors='replace').strip()
except Exception:
body_text = msg.get_payload()
# Use text if available, otherwise html
body = body_text if body_text else body_html
if not body:
body = "(No message body found)"
# Format the output
formatted_email = []
# Add headers in order
for header in headers_order:
if header in headers:
formatted_email.append(f"{header}: {headers[header]}")
if headers:
formatted_email.append("") # Blank line between headers and body
# Add body
formatted_email.append(body)
return '\n'.join(formatted_email)
except Exception as e:
logger.error(f"Error parsing email: {e}")
# Return original content if parsing fails
return email_content
# Initialize search index
def create_search_index():
"""Create or open the search index."""
schema = Schema(
path=ID(stored=True),
title=TEXT(stored=True),
content=TEXT(stored=False)
)
if not os.path.exists(os.path.join(INDEX_DIR, "WRITELOCK")):
return create_in(INDEX_DIR, schema)
return open_dir(INDEX_DIR)
ix = create_search_index()
def is_hidden_path(file_path):
"""Check if a file path contains any hidden directories or files (starting with dot)."""
path_obj = Path(file_path) if isinstance(file_path, str) else file_path
# Check all parts of the path including the filename
return any(part.startswith('.') for part in path_obj.parts)
def normalize_list_indentation(md_content):
"""
Normalize list indentation from 2-space to 4-space for proper markdown parsing.
Python-Markdown requires 4 spaces for nested lists, but many editors use 2 spaces.
Also ensures blank lines before lists when needed.
"""
lines = md_content.split('\n')
normalized_lines = []
def is_list_item_line(text):
"""Check if a line is actually a list item, not bold/italic markdown."""
stripped = text.lstrip(' ')
if not stripped:
return False
# Check for unordered list (- or + or single * followed by space)
if stripped[0] in '-+':
return len(stripped) > 1 and stripped[1] == ' '
elif stripped[0] == '*':
# Make sure it's not ** (bold) and is actually a list marker
return len(stripped) > 1 and stripped[1] == ' ' and not (len(stripped) > 2 and stripped[2] == '*')
# Check for ordered list (digits followed by . and space)
if stripped[0].isdigit():
i = 0
while i < len(stripped) and stripped[i].isdigit():
i += 1
return i < len(stripped) and stripped[i] == '.' and i + 1 < len(stripped) and stripped[i + 1] == ' '
return False
for i, line in enumerate(lines):
# Count leading spaces
leading_spaces = len(line) - len(line.lstrip(' '))
stripped = line.lstrip(' ')
is_list = is_list_item_line(line)
if is_list:
# Check if previous line exists and is not empty and is not a list item and is not a blank line
if i > 0 and normalized_lines:
prev_line = normalized_lines[-1]
prev_is_list = is_list_item_line(prev_line)
# If previous line exists, is not empty, and is not a list item, add blank line
if prev_line.strip() and not prev_is_list and prev_line != '':
normalized_lines.append('')
if leading_spaces > 0 and leading_spaces % 2 == 0:
# Convert 2-space indentation to 4-space for nested lists
# Each 2-space indent becomes 4-space
new_spaces = (leading_spaces // 2) * 4
normalized_lines.append(' ' * new_spaces + stripped)
else:
normalized_lines.append(line)
else:
normalized_lines.append(line)
return '\n'.join(normalized_lines)
def rewrite_image_paths(md_content, filepath):
"""Rewrite relative image paths to be served by Flask."""
# Get the directory of the current file
file_dir = os.path.dirname(filepath)
# Pattern to match markdown image syntax: ![alt](path)
def replace_image_path(match):
alt_text = match.group(1)
img_path = match.group(2)
# Skip absolute URLs and data URIs
if img_path.startswith(('http://', 'https://', 'data:')):
return match.group(0)
# Resolve relative paths
if img_path.startswith('/'):
# Absolute path from content root
resolved_path = img_path.lstrip('/')
else:
# Relative path - resolve it relative to the file's directory
if file_dir:
resolved_path = os.path.normpath(os.path.join(file_dir, img_path))
else:
resolved_path = img_path
# Ensure forward slashes for URL
resolved_path = resolved_path.replace(os.sep, '/')
image_url = f"/image/{resolved_path}"
return f"![{alt_text}]({image_url})"
# Replace all markdown image references
md_content = re.sub(r'!\[([^\]]*)\]\(([^\)]+)\)', replace_image_path, md_content)
return md_content
def rewrite_eml_file_links(md_content, filepath):
"""Rewrite relative .eml file links to be served by Flask's /file/ route."""
# Get the directory of the current file
file_dir = os.path.dirname(filepath)
# Pattern to match markdown link syntax: [text](path)
def replace_link_path(match):
link_text = match.group(1)
link_path = match.group(2)
# Only process .eml files
if not link_path.lower().endswith('.eml'):
return match.group(0)
# Skip absolute URLs
if link_path.startswith(('http://', 'https://', '/file/')):
return match.group(0)
# Resolve relative paths
if link_path.startswith('/'):
# Absolute path from content root
resolved_path = link_path.lstrip('/')
else:
# Relative path - resolve it relative to the file's directory
if file_dir:
resolved_path = os.path.normpath(os.path.join(file_dir, link_path))
else:
resolved_path = link_path
# Ensure forward slashes for URL
resolved_path = resolved_path.replace(os.sep, '/')
file_url = f"/file/{resolved_path}"
return f"[{link_text}]({file_url})"
# Replace all markdown link references to .eml files
md_content = re.sub(r'\[([^\]]+)\]\(([^\)]+\.eml)\)', replace_link_path, md_content)
return md_content
def build_directory_tree():
"""Build a hierarchical tree of markdown files organized by directory."""
tree = {}
if not os.path.exists(CONTENT_DIR):
logger.warning(f"Content directory does not exist: {CONTENT_DIR}")
return tree
file_count = 0
# Include both .md and .md.enc files
for md_file in Path(CONTENT_DIR).rglob('*'):
is_md = md_file.suffix.lower() in MARKDOWN_EXTENSIONS
is_enc = md_file.suffix == '.enc' and md_file.stem.endswith('.md')
if not (is_md or is_enc):
continue
if is_hidden_path(md_file):
logger.debug(f"Skipping hidden path: {md_file}")
continue
file_count += 1
rel_path = md_file.relative_to(CONTENT_DIR)
parts = rel_path.parts[:-1] # All parts except filename
# For encrypted files, show without .enc extension
if is_enc:
filename = md_file.stem # Removes .enc, keeping the .md
if filename.endswith('.md'):
filename = filename[:-3] # Remove .md to show clean name
else:
filename = md_file.stem # Name without extension
logger.debug(f"Adding to tree: {rel_path} (name: {filename})")
# Navigate/create nested dict structure
current = tree
for part in parts:
if part not in current:
current[part] = {}
current = current[part]
# Add file to current level
if '_files' not in current:
current['_files'] = []
# Check if file is protected
is_protected = is_file_protected(str(rel_path))
current['_files'].append({
'name': filename,
'path': str(rel_path),
'encrypted': is_enc,
'protected': is_protected
})
logger.info(f"Built tree for {CONTENT_DIR}: found {file_count} markdown files")
return tree
def update_search_index():
"""Update the search index with all markdown files."""
try:
writer = ix.writer()
for md_file in Path(CONTENT_DIR).rglob('*'):
if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS and not md_file.suffix == '.enc':
continue
if is_hidden_path(md_file):
continue
try:
# Handle encrypted files
rel_path = str(md_file.relative_to(CONTENT_DIR))
title = md_file.stem
if is_encrypted_file(rel_path):
# Skip encrypted files in search index (they need authentication)
logger.debug(f"Skipping encrypted file from search index: {rel_path}")
continue
content = md_file.read_text(encoding='utf-8', errors='ignore')
writer.add_document(
path=rel_path,
title=title,
content=content
)
except Exception as e:
logger.error(f"Error indexing {md_file}: {e}")
writer.commit()
logger.info("Search index updated successfully")
except Exception as e:
logger.error(f"Error updating search index: {e}")
# Routes
@app.route('/')
def index():
"""Display the main page with file navigation and search."""
file_tree = build_directory_tree()
return render_template('index.html', repositories=[{
'name': 'Content',
'tree': file_tree
}])
@app.route('/image/<path:filepath>')
def serve_image(filepath):
"""Serve images from content directory."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return "Invalid path", 400
file_path = os.path.join(CONTENT_DIR, filepath)
# Ensure the file is within the content directory
try:
file_path = os.path.realpath(file_path)
content_path = os.path.realpath(CONTENT_DIR)
if not file_path.startswith(content_path):
return "Access denied", 403
except Exception:
return "Invalid path", 400
if not os.path.exists(file_path):
logger.warning(f"Image not found: {file_path}")
return "Image not found", 404
try:
return send_file(file_path)
except Exception as e:
logger.error(f"Error serving image {file_path}: {e}")
return f"Error serving image: {e}", 500
@app.route('/file/<path:filepath>')
def serve_file(filepath):
"""Serve files from content directory, with base64 decoding for .eml files if needed."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return "Invalid path", 400
file_path = os.path.join(CONTENT_DIR, filepath)
# Ensure the file is within the content directory
try:
file_path = os.path.realpath(file_path)
content_path = os.path.realpath(CONTENT_DIR)
if not file_path.startswith(content_path):
return "Access denied", 403
except Exception:
return "Invalid path", 400
if not os.path.exists(file_path):
logger.warning(f"File not found: {file_path}")
return "File not found", 404
try:
# Check if this is an .eml file
if file_path.lower().endswith('.eml'):
with open(file_path, 'r', encoding='utf-8', errors='ignore') as f:
content = f.read()
# Check if content is base64 encoded
# Base64 content is ASCII and contains only base64 characters
try:
if content.strip() and all(c in 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r' for c in content):
decoded = base64.b64decode(content, validate=True).decode('utf-8', errors='ignore')
# If it looks like an email (has headers), use decoded version
if 'From:' in decoded or 'To:' in decoded or 'Subject:' in decoded:
content = decoded
except Exception:
# Not valid base64, use original content
pass
# Extract and format the email content
formatted_email = extract_email_text(content)
# Return as text/plain for viewing in browser
response = make_response(formatted_email)
response.headers['Content-Type'] = 'text/plain; charset=utf-8'
response.headers['Content-Disposition'] = f'inline; filename="{os.path.basename(file_path)}"'
return response
else:
# For other file types, just serve as-is
return send_file(file_path)
except Exception as e:
logger.error(f"Error serving file {file_path}: {e}")
return f"Error serving file: {e}", 500
@app.route('/api/search', methods=['GET'])
def search():
"""Search markdown files."""
query_str = request.args.get('q', '').strip()
if not query_str or len(query_str) < 2:
return jsonify({'results': []})
try:
with ix.searcher() as searcher:
query = QueryParser("content", ix.schema).parse(query_str)
results = searcher.search(query)
search_results = []
for hit in results:
search_results.append({
'path': hit['path'],
'title': hit['title']
})
return jsonify({'results': search_results})
except Exception as e:
logger.error(f"Search error: {e}")
return jsonify({'results': [], 'error': str(e)})
@app.route('/view/<path:filepath>')
def view_file(filepath):
"""View a markdown file converted to HTML."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return "Invalid path", 400
# Determine if looking for encrypted version
enc_filepath = filepath + '.enc' if not filepath.endswith('.enc') else filepath
# Try encrypted file first if it exists
file_path = os.path.join(CONTENT_DIR, enc_filepath)
is_encrypted = False
if os.path.exists(file_path):
is_encrypted = True
lookup_filepath = enc_filepath
else:
# Fall back to regular file
file_path = os.path.join(CONTENT_DIR, filepath)
lookup_filepath = filepath
# Ensure the file is within the content directory
try:
file_path = os.path.realpath(file_path)
content_path = os.path.realpath(CONTENT_DIR)
if not file_path.startswith(content_path):
return "Access denied", 403
except Exception:
return "Invalid path", 400
if not os.path.exists(file_path):
return "File not found", 404
# Check if file is protected and user is authenticated
if is_file_protected(lookup_filepath):
if not is_authenticated(lookup_filepath):
return render_template(
'password_prompt.html',
filepath=lookup_filepath,
filename=os.path.basename(filepath)
)
try:
# Read file content
if is_encrypted:
with open(file_path, 'r', encoding='utf-8') as f:
encrypted_content = f.read()
# Get password from session
file_key = get_file_key(lookup_filepath)
authenticated_files = session.get('authenticated_files', {})
password = authenticated_files.get(file_key + '_password')
if not password:
return "Unable to decrypt: password not found in session", 500
md_content = decrypt_file_content(encrypted_content, password)
if md_content is None:
return "Failed to decrypt file", 500
else:
with open(file_path, 'r', encoding='utf-8') as f:
md_content = f.read()
# Rewrite image paths to be served by Flask
md_content = rewrite_image_paths(md_content, filepath)
# Rewrite .eml file links to be served by Flask with base64 decoding
md_content = rewrite_eml_file_links(md_content, filepath)
# Normalize list indentation for proper markdown parsing
md_content = normalize_list_indentation(md_content)
# Convert markdown to HTML
html_content = markdown.markdown(
md_content,
extensions=['extra', 'codehilite', 'toc']
)
return render_template(
'view.html',
filepath=lookup_filepath,
content=html_content,
filename=os.path.basename(filepath),
is_protected=is_file_protected(lookup_filepath),
is_encrypted=is_encrypted
)
except Exception as e:
logger.error(f"Error reading file {file_path}: {e}")
return f"Error reading file: {e}", 500
@app.route('/api/auth/<path:filepath>', methods=['POST'])
def authenticate(filepath):
"""Authenticate user for a protected file."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return jsonify({'success': False, 'error': 'Invalid path'}), 400
password = request.form.get('password', '')
file_key = get_file_key(filepath)
passwords = load_passwords()
file_data = passwords.get(file_key)
if not file_data or not file_data.get('protected'):
return jsonify({'success': False, 'error': 'File not protected'}), 400
# Check password
if check_password_hash(file_data['password_hash'], password):
# Store in session
if 'authenticated_files' not in session:
session['authenticated_files'] = {}
session['authenticated_files'][file_key] = True
# Store the password for decryption if file is encrypted
if is_encrypted_file(filepath):
session['authenticated_files'][file_key + '_password'] = password
session.modified = True
logger.info(f"User authenticated for {file_key}")
return jsonify({'success': True, 'redirect': url_for('view_file', filepath=filepath)})
else:
logger.warning(f"Failed authentication attempt for {file_key}")
return jsonify({'success': False, 'error': 'Invalid password'}), 401
@app.route('/api/protect/<path:filepath>', methods=['POST'])
def protect_file(filepath):
"""Protect or unprotect a file with a password."""
# This should be restricted to admin users in production
# For now, requires a master password via environment variable
master_password = os.environ.get('MARKMYWORDS_ADMIN_PASSWORD')
auth_header = request.headers.get('Authorization', '')
if not auth_header.startswith('Bearer '):
return jsonify({'success': False, 'error': 'Missing authorization'}), 401
token = auth_header.split(' ')[1]
if not master_password or token != master_password:
return jsonify({'success': False, 'error': 'Invalid authorization'}), 401
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return jsonify({'success': False, 'error': 'Invalid path'}), 400
action = request.json.get('action') # 'protect' or 'unprotect'
new_password = request.json.get('password')
encrypt_file = request.json.get('encrypt_file', False) # Whether to encrypt the file
file_key = get_file_key(filepath)
passwords = load_passwords()
if action == 'protect':
if not new_password:
return jsonify({'success': False, 'error': 'Password required'}), 400
passwords[file_key] = {
'protected': True,
'password_hash': generate_password_hash(new_password),
'created_at': datetime.now().isoformat(),
'encrypted': encrypt_file
}
# If encryption is requested, encrypt the file
if encrypt_file and ENCRYPTION_ENABLED:
try:
file_path = os.path.join(CONTENT_DIR, filepath)
if os.path.exists(file_path):
with open(file_path, 'r', encoding='utf-8') as f:
original_content = f.read()
encrypted_content = encrypt_file_content(original_content, new_password)
if encrypted_content:
# Save encrypted file with .enc extension
enc_file_path = file_path + '.enc'
with open(enc_file_path, 'w', encoding='utf-8') as f:
f.write(encrypted_content)
# Delete original unencrypted file
os.remove(file_path)
logger.info(f"Encrypted file: {filepath}")
except Exception as e:
logger.error(f"Error encrypting file {filepath}: {e}")
return jsonify({'success': False, 'error': f"Failed to encrypt file: {e}"}), 500
logger.info(f"Protected file: {file_key}")
elif action == 'unprotect':
if file_key in passwords:
del passwords[file_key]
logger.info(f"Unprotected file: {file_key}")
else:
return jsonify({'success': False, 'error': 'Invalid action'}), 400
if save_passwords(passwords):
return jsonify({'success': True, 'message': f"File {action}ed successfully"})
else:
return jsonify({'success': False, 'error': 'Failed to save password'}), 500
@app.route('/api/status')
def status():
"""Return application status."""
content_status = {
'exists': os.path.exists(CONTENT_DIR),
'last_modified': datetime.fromtimestamp(
os.path.getmtime(CONTENT_DIR)
).isoformat() if os.path.exists(CONTENT_DIR) else None
}
return jsonify({
'status': 'running',
'content': content_status,
'encryption_enabled': ENCRYPTION_ENABLED,
'timestamp': datetime.now().isoformat()
})
def setup_scheduler():
"""Set up background scheduler for periodic content scanning."""
scheduler = BackgroundScheduler()
scheduler.add_job(
func=update_search_index,
trigger=IntervalTrigger(seconds=60),
id='update_search_index',
name='Update search index every 60 seconds',
replace_existing=True
)
scheduler.start()
logger.info("Scheduler started: content directory will be scanned every 60 seconds")
return scheduler
if __name__ == '__main__':
# Perform initial index build
logger.info("Initializing markmywords")
update_search_index()
# Start background scheduler for periodic content scanning
scheduler = setup_scheduler()
try:
# Start Flask app
app.run(host='0.0.0.0', port=5000, debug=False)
finally:
# Shut down scheduler on exit
scheduler.shutdown()