#!/usr/bin/env python3 """ markmywords - Self-hosted markdown repository viewer with search """ import os import subprocess import json from pathlib import Path from datetime import datetime from flask import Flask, render_template, request, jsonify, send_file from apscheduler.schedulers.background import BackgroundScheduler from apscheduler.triggers.cron import CronTrigger import markdown import logging import re from whoosh.index import create_in, open_dir from whoosh.fields import Schema, TEXT, ID from whoosh.qparser import QueryParser app = Flask(__name__) # Configuration CONFIG_FILE = "/config/repositories.json" REPOS_DIR = "/data/repos" INDEX_DIR = "/data/search_index" MARKDOWN_EXTENSIONS = ['.md', '.markdown'] # Setup logging logging.basicConfig(level=logging.INFO) logger = logging.getLogger(__name__) # Ensure directories exist os.makedirs(REPOS_DIR, exist_ok=True) os.makedirs(INDEX_DIR, exist_ok=True) # Initialize search index def create_search_index(): """Create or open the search index.""" schema = Schema( path=ID(stored=True), title=TEXT(stored=True), content=TEXT(stored=False) ) if not os.path.exists(os.path.join(INDEX_DIR, "WRITELOCK")): return create_in(INDEX_DIR, schema) return open_dir(INDEX_DIR) ix = create_search_index() def is_hidden_path(file_path): """Check if a file path contains any hidden directories or files (starting with dot).""" path_obj = Path(file_path) if isinstance(file_path, str) else file_path # Check all parts of the path including the filename return any(part.startswith('.') for part in path_obj.parts) def rewrite_image_paths(md_content, repo, filepath): """Rewrite relative image paths to be served by Flask.""" # Get the directory of the current file (relative to repo root) file_dir = os.path.dirname(filepath) # Pattern to match markdown image syntax: ![alt](path) def replace_image_path(match): alt_text = match.group(1) img_path = match.group(2) # Skip absolute URLs and data URIs if img_path.startswith(('http://', 'https://', 'data:')): return match.group(0) # Resolve relative paths if img_path.startswith('/'): # Absolute path from repo root resolved_path = img_path.lstrip('/') else: # Relative path - resolve it relative to the file's directory if file_dir: resolved_path = os.path.normpath(os.path.join(file_dir, img_path)) else: resolved_path = img_path # Ensure forward slashes for URL resolved_path = resolved_path.replace(os.sep, '/') image_url = f"/image/{repo}/{resolved_path}" return f"![{alt_text}]({image_url})" # Replace all markdown image references md_content = re.sub(r'!\[([^\]]*)\]\(([^\)]+)\)', replace_image_path, md_content) return md_content def build_directory_tree(repo_path): """Build a hierarchical tree of markdown files organized by directory.""" tree = {} if not os.path.exists(repo_path): logger.warning(f"Repo path does not exist: {repo_path}") return tree file_count = 0 for md_file in Path(repo_path).rglob('*'): if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS: continue if is_hidden_path(md_file): logger.debug(f"Skipping hidden path: {md_file}") continue file_count += 1 rel_path = md_file.relative_to(repo_path) parts = rel_path.parts[:-1] # All parts except filename filename = md_file.stem # Name without extension logger.debug(f"Adding to tree: {rel_path} (name: {filename})") # Navigate/create nested dict structure current = tree for part in parts: if part not in current: current[part] = {} current = current[part] # Add file to current level if '_files' not in current: current['_files'] = [] current['_files'].append({ 'name': filename, 'path': str(rel_path) }) logger.info(f"Built tree for {repo_path}: found {file_count} markdown files") return tree def load_repositories(): """Load repository configuration from file.""" if not os.path.exists(CONFIG_FILE): logger.warning(f"Config file not found: {CONFIG_FILE}") return {} try: with open(CONFIG_FILE, 'r') as f: return json.load(f) except Exception as e: logger.error(f"Error loading config: {e}") return {} def git_pull_repo(repo_name, repo_path): """Perform a git pull on the specified repository.""" try: logger.info(f"Pulling repository: {repo_name}") result = subprocess.run( ["git", "pull"], cwd=repo_path, capture_output=True, text=True, timeout=300 ) logger.info(f"Pull result for {repo_name}: {result.stdout}") if result.returncode != 0: logger.error(f"Pull error for {repo_name}: {result.stderr}") return result.returncode == 0 except subprocess.TimeoutExpired: logger.error(f"Git pull timeout for {repo_name}") return False except Exception as e: logger.error(f"Error pulling {repo_name}: {e}") return False def clone_or_pull(repo_name, repo_url): """Clone repository if it doesn't exist, otherwise pull.""" repo_path = os.path.join(REPOS_DIR, repo_name) if not os.path.exists(repo_path): try: logger.info(f"Cloning repository: {repo_name} from {repo_url}") subprocess.run( ["git", "clone", repo_url, repo_path], capture_output=True, text=True, timeout=300 ) logger.info(f"Successfully cloned {repo_name}") except Exception as e: logger.error(f"Error cloning {repo_name}: {e}") return False else: return git_pull_repo(repo_name, repo_path) return True def update_search_index(): """Update the search index with all markdown files.""" try: writer = ix.writer() for repo_dir in Path(REPOS_DIR).iterdir(): if not repo_dir.is_dir(): continue for md_file in repo_dir.rglob('*'): if md_file.suffix.lower() in MARKDOWN_EXTENSIONS and not is_hidden_path(md_file): try: content = md_file.read_text(encoding='utf-8', errors='ignore') # Extract title from filename or first heading title = md_file.stem rel_path = str(md_file.relative_to(REPOS_DIR)) writer.add_document( path=rel_path, title=title, content=content ) except Exception as e: logger.error(f"Error indexing {md_file}: {e}") writer.commit() logger.info("Search index updated successfully") except Exception as e: logger.error(f"Error updating search index: {e}") def scheduled_pull(): """Perform scheduled pulls of all repositories.""" logger.info("Starting scheduled repository pull") repos = load_repositories() for repo_name, repo_config in repos.items(): if not repo_config.get('enabled', True): continue clone_or_pull(repo_name, repo_config['url']) # Update search index after pulling update_search_index() logger.info("Scheduled pull completed") def setup_scheduler(): """Setup the background scheduler for periodic pulls.""" scheduler = BackgroundScheduler() repos = load_repositories() for repo_name, repo_config in repos.items(): if repo_config.get('enabled', True): cron_schedule = repo_config.get('schedule', '0 */6 * * *') # Default: every 6 hours try: scheduler.add_job( scheduled_pull, CronTrigger.from_crontab(cron_schedule), id=f"pull_{repo_name}", name=f"Pull {repo_name}" ) logger.info(f"Scheduled pull for {repo_name}: {cron_schedule}") except Exception as e: logger.error(f"Error scheduling {repo_name}: {e}") scheduler.start() return scheduler # Routes @app.route('/') def index(): """Display the main page with list of repositories and search.""" repos = load_repositories() repo_list = [] for repo_name in repos.keys(): repo_path = os.path.join(REPOS_DIR, repo_name) file_tree = build_directory_tree(repo_path) repo_list.append({ 'name': repo_name, 'tree': file_tree }) return render_template('index.html', repositories=repo_list) @app.route('/image//') def serve_image(repo, filepath): """Serve images from repository directories.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return "Invalid path", 400 file_path = os.path.join(REPOS_DIR, repo, filepath) # Ensure the file is within the repo directory try: file_path = os.path.realpath(file_path) repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo)) if not file_path.startswith(repo_path): return "Access denied", 403 except Exception: return "Invalid path", 400 if not os.path.exists(file_path): logger.warning(f"Image not found: {file_path}") return "Image not found", 404 try: return send_file(file_path) except Exception as e: logger.error(f"Error serving image {file_path}: {e}") return f"Error serving image: {e}", 500 @app.route('/api/search', methods=['GET']) def search(): """Search markdown files.""" query_str = request.args.get('q', '').strip() if not query_str or len(query_str) < 2: return jsonify({'results': []}) try: with ix.searcher() as searcher: query = QueryParser("content", ix.schema).parse(query_str) results = searcher.search(query) search_results = [] for hit in results: search_results.append({ 'path': hit['path'], 'title': hit['title'] }) return jsonify({'results': search_results}) except Exception as e: logger.error(f"Search error: {e}") return jsonify({'results': [], 'error': str(e)}) @app.route('/view//') def view_file(repo, filepath): """View a markdown file converted to HTML.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return "Invalid path", 400 file_path = os.path.join(REPOS_DIR, repo, filepath) # Ensure the file is within the repo directory try: file_path = os.path.realpath(file_path) repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo)) if not file_path.startswith(repo_path): return "Access denied", 403 except Exception: return "Invalid path", 400 if not os.path.exists(file_path): return "File not found", 404 try: with open(file_path, 'r', encoding='utf-8') as f: md_content = f.read() # Rewrite image paths to be served by Flask md_content = rewrite_image_paths(md_content, repo, filepath) # Convert markdown to HTML html_content = markdown.markdown( md_content, extensions=['extra', 'codehilite', 'toc'] ) return render_template( 'view.html', repo=repo, filepath=filepath, content=html_content, filename=os.path.basename(filepath) ) except Exception as e: logger.error(f"Error reading file {file_path}: {e}") return f"Error reading file: {e}", 500 @app.route('/api/status') def status(): """Return application status.""" repos = load_repositories() repo_status = {} for repo_name in repos.keys(): repo_path = os.path.join(REPOS_DIR, repo_name) repo_status[repo_name] = { 'cloned': os.path.exists(repo_path), 'last_modified': datetime.fromtimestamp( os.path.getmtime(repo_path) ).isoformat() if os.path.exists(repo_path) else None } return jsonify({ 'status': 'running', 'repositories': repo_status, 'timestamp': datetime.now().isoformat() }) if __name__ == '__main__': # Perform initial pull and index logger.info("Initializing markmywords") scheduled_pull() # Setup scheduler scheduler = setup_scheduler() # Start Flask app app.run(host='0.0.0.0', port=5000, debug=False)