#!/usr/bin/env python3 """ markmywords - Self-hosted markdown viewer with search and encryption """ import os import json from pathlib import Path from datetime import datetime from flask import Flask, render_template, request, jsonify, send_file, session, redirect, url_for, make_response from werkzeug.security import generate_password_hash, check_password_hash import markdown import logging import re from whoosh.index import create_in, open_dir from whoosh.fields import Schema, TEXT, ID from whoosh.qparser import QueryParser from cryptography.fernet import Fernet from cryptography.hazmat.primitives import hashes from cryptography.hazmat.primitives.kdf.pbkdf2 import PBKDF2HMAC from cryptography.hazmat.backends import default_backend import base64 from apscheduler.schedulers.background import BackgroundScheduler from apscheduler.triggers.interval import IntervalTrigger app = Flask(__name__) app.secret_key = os.environ.get('FLASK_SECRET_KEY', 'dev-secret-key-change-in-production') # Configuration CONTENT_DIR = "/content" INDEX_DIR = "/data/search_index" PASSWORD_FILE = "/config/page_passwords.json" MARKDOWN_EXTENSIONS = ['.md', '.markdown'] ENCRYPTION_ENABLED = os.environ.get('ENCRYPTION_ENABLED', 'true').lower() == 'true' # Setup logging logging.basicConfig(level=logging.INFO) logger = logging.getLogger(__name__) # Ensure directories exist os.makedirs(CONTENT_DIR, exist_ok=True) os.makedirs(INDEX_DIR, exist_ok=True) os.makedirs(os.path.dirname(PASSWORD_FILE), exist_ok=True) def load_passwords(): """Load password database from file.""" if not os.path.exists(PASSWORD_FILE): return {} try: with open(PASSWORD_FILE, 'r') as f: return json.load(f) except Exception as e: logger.error(f"Error loading passwords: {e}") return {} def save_passwords(passwords): """Save password database to file.""" try: os.makedirs(os.path.dirname(PASSWORD_FILE), exist_ok=True) with open(PASSWORD_FILE, 'w') as f: json.dump(passwords, f, indent=2) os.chmod(PASSWORD_FILE, 0o600) # Restrict file permissions return True except Exception as e: logger.error(f"Error saving passwords: {e}") return False def derive_encryption_key(password): """Derive an encryption key from a password.""" salt = b'markmywords_salt' # Fixed salt for consistency kdf = PBKDF2HMAC( algorithm=hashes.SHA256(), length=32, salt=salt, iterations=100000, backend=default_backend() ) key = base64.urlsafe_b64encode(kdf.derive(password.encode())) return key def encrypt_file_content(content, password): """Encrypt file content using password-derived key.""" try: key = derive_encryption_key(password) f = Fernet(key) encrypted = f.encrypt(content.encode('utf-8')) return encrypted.decode('utf-8') except Exception as e: logger.error(f"Error encrypting content: {e}") return None def decrypt_file_content(encrypted_content, password): """Decrypt file content using password-derived key.""" try: key = derive_encryption_key(password) f = Fernet(key) decrypted = f.decrypt(encrypted_content.encode('utf-8')) return decrypted.decode('utf-8') except Exception as e: logger.error(f"Error decrypting content: {e}") return None def is_encrypted_file(filepath): """Check if a file is encrypted (has .enc extension).""" return filepath.endswith('.enc') def get_file_key(filepath): """Generate a consistent key for a file.""" return filepath def is_file_protected(filepath): """Check if a file is password protected.""" passwords = load_passwords() file_key = get_file_key(filepath) return passwords.get(file_key, {}).get('protected', False) def is_authenticated(filepath): """Check if user is authenticated for a protected file.""" if not is_file_protected(filepath): return True # Not protected, so allowed file_key = get_file_key(filepath) authenticated_files = session.get('authenticated_files', {}) return authenticated_files.get(file_key, False) # Initialize search index def create_search_index(): """Create or open the search index.""" schema = Schema( path=ID(stored=True), title=TEXT(stored=True), content=TEXT(stored=False) ) if not os.path.exists(os.path.join(INDEX_DIR, "WRITELOCK")): return create_in(INDEX_DIR, schema) return open_dir(INDEX_DIR) ix = create_search_index() def is_hidden_path(file_path): """Check if a file path contains any hidden directories or files (starting with dot).""" path_obj = Path(file_path) if isinstance(file_path, str) else file_path # Check all parts of the path including the filename return any(part.startswith('.') for part in path_obj.parts) def normalize_list_indentation(md_content): """ Normalize list indentation from 2-space to 4-space for proper markdown parsing. Python-Markdown requires 4 spaces for nested lists, but many editors use 2 spaces. Also ensures blank lines before lists when needed. """ lines = md_content.split('\n') normalized_lines = [] def is_list_item_line(text): """Check if a line is actually a list item, not bold/italic markdown.""" stripped = text.lstrip(' ') if not stripped: return False # Check for unordered list (- or + or single * followed by space) if stripped[0] in '-+': return len(stripped) > 1 and stripped[1] == ' ' elif stripped[0] == '*': # Make sure it's not ** (bold) and is actually a list marker return len(stripped) > 1 and stripped[1] == ' ' and not (len(stripped) > 2 and stripped[2] == '*') # Check for ordered list (digits followed by . and space) if stripped[0].isdigit(): i = 0 while i < len(stripped) and stripped[i].isdigit(): i += 1 return i < len(stripped) and stripped[i] == '.' and i + 1 < len(stripped) and stripped[i + 1] == ' ' return False for i, line in enumerate(lines): # Count leading spaces leading_spaces = len(line) - len(line.lstrip(' ')) stripped = line.lstrip(' ') is_list = is_list_item_line(line) if is_list: # Check if previous line exists and is not empty and is not a list item and is not a blank line if i > 0 and normalized_lines: prev_line = normalized_lines[-1] prev_is_list = is_list_item_line(prev_line) # If previous line exists, is not empty, and is not a list item, add blank line if prev_line.strip() and not prev_is_list and prev_line != '': normalized_lines.append('') if leading_spaces > 0 and leading_spaces % 2 == 0: # Convert 2-space indentation to 4-space for nested lists # Each 2-space indent becomes 4-space new_spaces = (leading_spaces // 2) * 4 normalized_lines.append(' ' * new_spaces + stripped) else: normalized_lines.append(line) else: normalized_lines.append(line) return '\n'.join(normalized_lines) def rewrite_image_paths(md_content, filepath): """Rewrite relative image paths to be served by Flask.""" # Get the directory of the current file file_dir = os.path.dirname(filepath) # Pattern to match markdown image syntax: ![alt](path) def replace_image_path(match): alt_text = match.group(1) img_path = match.group(2) # Skip absolute URLs and data URIs if img_path.startswith(('http://', 'https://', 'data:')): return match.group(0) # Resolve relative paths if img_path.startswith('/'): # Absolute path from content root resolved_path = img_path.lstrip('/') else: # Relative path - resolve it relative to the file's directory if file_dir: resolved_path = os.path.normpath(os.path.join(file_dir, img_path)) else: resolved_path = img_path # Ensure forward slashes for URL resolved_path = resolved_path.replace(os.sep, '/') image_url = f"/image/{resolved_path}" return f"![{alt_text}]({image_url})" # Replace all markdown image references md_content = re.sub(r'!\[([^\]]*)\]\(([^\)]+)\)', replace_image_path, md_content) return md_content def rewrite_eml_file_links(md_content, filepath): """Rewrite relative .eml file links to be served by Flask's /file/ route.""" # Get the directory of the current file file_dir = os.path.dirname(filepath) # Pattern to match markdown link syntax: [text](path) def replace_link_path(match): link_text = match.group(1) link_path = match.group(2) # Only process .eml files if not link_path.lower().endswith('.eml'): return match.group(0) # Skip absolute URLs if link_path.startswith(('http://', 'https://', '/file/')): return match.group(0) # Resolve relative paths if link_path.startswith('/'): # Absolute path from content root resolved_path = link_path.lstrip('/') else: # Relative path - resolve it relative to the file's directory if file_dir: resolved_path = os.path.normpath(os.path.join(file_dir, link_path)) else: resolved_path = link_path # Ensure forward slashes for URL resolved_path = resolved_path.replace(os.sep, '/') file_url = f"/file/{resolved_path}" return f"[{link_text}]({file_url})" # Replace all markdown link references to .eml files md_content = re.sub(r'\[([^\]]+)\]\(([^\)]+\.eml)\)', replace_link_path, md_content) return md_content def build_directory_tree(): """Build a hierarchical tree of markdown files organized by directory.""" tree = {} if not os.path.exists(CONTENT_DIR): logger.warning(f"Content directory does not exist: {CONTENT_DIR}") return tree file_count = 0 # Include both .md and .md.enc files for md_file in Path(CONTENT_DIR).rglob('*'): is_md = md_file.suffix.lower() in MARKDOWN_EXTENSIONS is_enc = md_file.suffix == '.enc' and md_file.stem.endswith('.md') if not (is_md or is_enc): continue if is_hidden_path(md_file): logger.debug(f"Skipping hidden path: {md_file}") continue file_count += 1 rel_path = md_file.relative_to(CONTENT_DIR) parts = rel_path.parts[:-1] # All parts except filename # For encrypted files, show without .enc extension if is_enc: filename = md_file.stem # Removes .enc, keeping the .md if filename.endswith('.md'): filename = filename[:-3] # Remove .md to show clean name else: filename = md_file.stem # Name without extension logger.debug(f"Adding to tree: {rel_path} (name: {filename})") # Navigate/create nested dict structure current = tree for part in parts: if part not in current: current[part] = {} current = current[part] # Add file to current level if '_files' not in current: current['_files'] = [] # Check if file is protected is_protected = is_file_protected(str(rel_path)) current['_files'].append({ 'name': filename, 'path': str(rel_path), 'encrypted': is_enc, 'protected': is_protected }) logger.info(f"Built tree for {CONTENT_DIR}: found {file_count} markdown files") return tree def update_search_index(): """Update the search index with all markdown files.""" try: writer = ix.writer() for md_file in Path(CONTENT_DIR).rglob('*'): if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS and not md_file.suffix == '.enc': continue if is_hidden_path(md_file): continue try: # Handle encrypted files rel_path = str(md_file.relative_to(CONTENT_DIR)) title = md_file.stem if is_encrypted_file(rel_path): # Skip encrypted files in search index (they need authentication) logger.debug(f"Skipping encrypted file from search index: {rel_path}") continue content = md_file.read_text(encoding='utf-8', errors='ignore') writer.add_document( path=rel_path, title=title, content=content ) except Exception as e: logger.error(f"Error indexing {md_file}: {e}") writer.commit() logger.info("Search index updated successfully") except Exception as e: logger.error(f"Error updating search index: {e}") # Routes @app.route('/') def index(): """Display the main page with file navigation and search.""" file_tree = build_directory_tree() return render_template('index.html', repositories=[{ 'name': 'Content', 'tree': file_tree }]) @app.route('/image/') def serve_image(filepath): """Serve images from content directory.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return "Invalid path", 400 file_path = os.path.join(CONTENT_DIR, filepath) # Ensure the file is within the content directory try: file_path = os.path.realpath(file_path) content_path = os.path.realpath(CONTENT_DIR) if not file_path.startswith(content_path): return "Access denied", 403 except Exception: return "Invalid path", 400 if not os.path.exists(file_path): logger.warning(f"Image not found: {file_path}") return "Image not found", 404 try: return send_file(file_path) except Exception as e: logger.error(f"Error serving image {file_path}: {e}") return f"Error serving image: {e}", 500 @app.route('/file/') def serve_file(filepath): """Serve files from content directory, with base64 decoding for .eml files if needed.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return "Invalid path", 400 file_path = os.path.join(CONTENT_DIR, filepath) # Ensure the file is within the content directory try: file_path = os.path.realpath(file_path) content_path = os.path.realpath(CONTENT_DIR) if not file_path.startswith(content_path): return "Access denied", 403 except Exception: return "Invalid path", 400 if not os.path.exists(file_path): logger.warning(f"File not found: {file_path}") return "File not found", 404 try: # Check if this is an .eml file if file_path.lower().endswith('.eml'): with open(file_path, 'r', encoding='utf-8', errors='ignore') as f: content = f.read() # Check if content is base64 encoded # Base64 content is ASCII and contains only base64 characters is_base64 = False try: if content.strip() and all(c in 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r' for c in content): decoded = base64.b64decode(content, validate=True).decode('utf-8', errors='ignore') # If it looks like an email (has headers), use decoded version if 'From:' in decoded or 'To:' in decoded or 'Subject:' in decoded: content = decoded is_base64 = True except Exception: # Not valid base64, use original content pass # Return as text/plain for viewing in browser response = make_response(content) response.headers['Content-Type'] = 'text/plain; charset=utf-8' response.headers['Content-Disposition'] = f'inline; filename="{os.path.basename(file_path)}"' return response else: # For other file types, just serve as-is return send_file(file_path) except Exception as e: logger.error(f"Error serving file {file_path}: {e}") return f"Error serving file: {e}", 500 @app.route('/api/search', methods=['GET']) def search(): """Search markdown files.""" query_str = request.args.get('q', '').strip() if not query_str or len(query_str) < 2: return jsonify({'results': []}) try: with ix.searcher() as searcher: query = QueryParser("content", ix.schema).parse(query_str) results = searcher.search(query) search_results = [] for hit in results: search_results.append({ 'path': hit['path'], 'title': hit['title'] }) return jsonify({'results': search_results}) except Exception as e: logger.error(f"Search error: {e}") return jsonify({'results': [], 'error': str(e)}) @app.route('/view/') def view_file(filepath): """View a markdown file converted to HTML.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return "Invalid path", 400 # Determine if looking for encrypted version enc_filepath = filepath + '.enc' if not filepath.endswith('.enc') else filepath # Try encrypted file first if it exists file_path = os.path.join(CONTENT_DIR, enc_filepath) is_encrypted = False if os.path.exists(file_path): is_encrypted = True lookup_filepath = enc_filepath else: # Fall back to regular file file_path = os.path.join(CONTENT_DIR, filepath) lookup_filepath = filepath # Ensure the file is within the content directory try: file_path = os.path.realpath(file_path) content_path = os.path.realpath(CONTENT_DIR) if not file_path.startswith(content_path): return "Access denied", 403 except Exception: return "Invalid path", 400 if not os.path.exists(file_path): return "File not found", 404 # Check if file is protected and user is authenticated if is_file_protected(lookup_filepath): if not is_authenticated(lookup_filepath): return render_template( 'password_prompt.html', filepath=lookup_filepath, filename=os.path.basename(filepath) ) try: # Read file content if is_encrypted: with open(file_path, 'r', encoding='utf-8') as f: encrypted_content = f.read() # Get password from session file_key = get_file_key(lookup_filepath) authenticated_files = session.get('authenticated_files', {}) password = authenticated_files.get(file_key + '_password') if not password: return "Unable to decrypt: password not found in session", 500 md_content = decrypt_file_content(encrypted_content, password) if md_content is None: return "Failed to decrypt file", 500 else: with open(file_path, 'r', encoding='utf-8') as f: md_content = f.read() # Rewrite image paths to be served by Flask md_content = rewrite_image_paths(md_content, filepath) # Rewrite .eml file links to be served by Flask with base64 decoding md_content = rewrite_eml_file_links(md_content, filepath) # Normalize list indentation for proper markdown parsing md_content = normalize_list_indentation(md_content) # Convert markdown to HTML html_content = markdown.markdown( md_content, extensions=['extra', 'codehilite', 'toc'] ) return render_template( 'view.html', filepath=lookup_filepath, content=html_content, filename=os.path.basename(filepath), is_protected=is_file_protected(lookup_filepath), is_encrypted=is_encrypted ) except Exception as e: logger.error(f"Error reading file {file_path}: {e}") return f"Error reading file: {e}", 500 @app.route('/api/auth/', methods=['POST']) def authenticate(filepath): """Authenticate user for a protected file.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return jsonify({'success': False, 'error': 'Invalid path'}), 400 password = request.form.get('password', '') file_key = get_file_key(filepath) passwords = load_passwords() file_data = passwords.get(file_key) if not file_data or not file_data.get('protected'): return jsonify({'success': False, 'error': 'File not protected'}), 400 # Check password if check_password_hash(file_data['password_hash'], password): # Store in session if 'authenticated_files' not in session: session['authenticated_files'] = {} session['authenticated_files'][file_key] = True # Store the password for decryption if file is encrypted if is_encrypted_file(filepath): session['authenticated_files'][file_key + '_password'] = password session.modified = True logger.info(f"User authenticated for {file_key}") return jsonify({'success': True, 'redirect': url_for('view_file', filepath=filepath)}) else: logger.warning(f"Failed authentication attempt for {file_key}") return jsonify({'success': False, 'error': 'Invalid password'}), 401 @app.route('/api/protect/', methods=['POST']) def protect_file(filepath): """Protect or unprotect a file with a password.""" # This should be restricted to admin users in production # For now, requires a master password via environment variable master_password = os.environ.get('MARKMYWORDS_ADMIN_PASSWORD') auth_header = request.headers.get('Authorization', '') if not auth_header.startswith('Bearer '): return jsonify({'success': False, 'error': 'Missing authorization'}), 401 token = auth_header.split(' ')[1] if not master_password or token != master_password: return jsonify({'success': False, 'error': 'Invalid authorization'}), 401 # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return jsonify({'success': False, 'error': 'Invalid path'}), 400 action = request.json.get('action') # 'protect' or 'unprotect' new_password = request.json.get('password') encrypt_file = request.json.get('encrypt_file', False) # Whether to encrypt the file file_key = get_file_key(filepath) passwords = load_passwords() if action == 'protect': if not new_password: return jsonify({'success': False, 'error': 'Password required'}), 400 passwords[file_key] = { 'protected': True, 'password_hash': generate_password_hash(new_password), 'created_at': datetime.now().isoformat(), 'encrypted': encrypt_file } # If encryption is requested, encrypt the file if encrypt_file and ENCRYPTION_ENABLED: try: file_path = os.path.join(CONTENT_DIR, filepath) if os.path.exists(file_path): with open(file_path, 'r', encoding='utf-8') as f: original_content = f.read() encrypted_content = encrypt_file_content(original_content, new_password) if encrypted_content: # Save encrypted file with .enc extension enc_file_path = file_path + '.enc' with open(enc_file_path, 'w', encoding='utf-8') as f: f.write(encrypted_content) # Delete original unencrypted file os.remove(file_path) logger.info(f"Encrypted file: {filepath}") except Exception as e: logger.error(f"Error encrypting file {filepath}: {e}") return jsonify({'success': False, 'error': f"Failed to encrypt file: {e}"}), 500 logger.info(f"Protected file: {file_key}") elif action == 'unprotect': if file_key in passwords: del passwords[file_key] logger.info(f"Unprotected file: {file_key}") else: return jsonify({'success': False, 'error': 'Invalid action'}), 400 if save_passwords(passwords): return jsonify({'success': True, 'message': f"File {action}ed successfully"}) else: return jsonify({'success': False, 'error': 'Failed to save password'}), 500 @app.route('/api/status') def status(): """Return application status.""" content_status = { 'exists': os.path.exists(CONTENT_DIR), 'last_modified': datetime.fromtimestamp( os.path.getmtime(CONTENT_DIR) ).isoformat() if os.path.exists(CONTENT_DIR) else None } return jsonify({ 'status': 'running', 'content': content_status, 'encryption_enabled': ENCRYPTION_ENABLED, 'timestamp': datetime.now().isoformat() }) def setup_scheduler(): """Set up background scheduler for periodic content scanning.""" scheduler = BackgroundScheduler() scheduler.add_job( func=update_search_index, trigger=IntervalTrigger(seconds=60), id='update_search_index', name='Update search index every 60 seconds', replace_existing=True ) scheduler.start() logger.info("Scheduler started: content directory will be scanned every 60 seconds") return scheduler if __name__ == '__main__': # Perform initial index build logger.info("Initializing markmywords") update_search_index() # Start background scheduler for periodic content scanning scheduler = setup_scheduler() try: # Start Flask app app.run(host='0.0.0.0', port=5000, debug=False) finally: # Shut down scheduler on exit scheduler.shutdown()