#!/usr/bin/env python3 """ markmywords - Self-hosted markdown repository viewer with search """ import os import subprocess import json from pathlib import Path from datetime import datetime from flask import Flask, render_template, request, jsonify from apscheduler.schedulers.background import BackgroundScheduler from apscheduler.triggers.cron import CronTrigger import markdown import logging from whoosh.index import create_in, open_dir from whoosh.fields import Schema, TEXT, ID from whoosh.qparser import QueryParser app = Flask(__name__) # Configuration CONFIG_FILE = "/config/repositories.json" REPOS_DIR = "/data/repos" INDEX_DIR = "/data/search_index" MARKDOWN_EXTENSIONS = ['.md', '.markdown'] # Setup logging logging.basicConfig(level=logging.INFO) logger = logging.getLogger(__name__) # Ensure directories exist os.makedirs(REPOS_DIR, exist_ok=True) os.makedirs(INDEX_DIR, exist_ok=True) # Initialize search index def create_search_index(): """Create or open the search index.""" schema = Schema( path=ID(stored=True), title=TEXT(stored=True), content=TEXT(stored=False) ) if not os.path.exists(os.path.join(INDEX_DIR, "WRITELOCK")): return create_in(INDEX_DIR, schema) return open_dir(INDEX_DIR) ix = create_search_index() def load_repositories(): """Load repository configuration from file.""" if not os.path.exists(CONFIG_FILE): logger.warning(f"Config file not found: {CONFIG_FILE}") return {} try: with open(CONFIG_FILE, 'r') as f: return json.load(f) except Exception as e: logger.error(f"Error loading config: {e}") return {} def git_pull_repo(repo_name, repo_path): """Perform a git pull on the specified repository.""" try: logger.info(f"Pulling repository: {repo_name}") result = subprocess.run( ["git", "pull"], cwd=repo_path, capture_output=True, text=True, timeout=300 ) logger.info(f"Pull result for {repo_name}: {result.stdout}") if result.returncode != 0: logger.error(f"Pull error for {repo_name}: {result.stderr}") return result.returncode == 0 except subprocess.TimeoutExpired: logger.error(f"Git pull timeout for {repo_name}") return False except Exception as e: logger.error(f"Error pulling {repo_name}: {e}") return False def clone_or_pull(repo_name, repo_url): """Clone repository if it doesn't exist, otherwise pull.""" repo_path = os.path.join(REPOS_DIR, repo_name) if not os.path.exists(repo_path): try: logger.info(f"Cloning repository: {repo_name} from {repo_url}") subprocess.run( ["git", "clone", repo_url, repo_path], capture_output=True, text=True, timeout=300 ) logger.info(f"Successfully cloned {repo_name}") except Exception as e: logger.error(f"Error cloning {repo_name}: {e}") return False else: return git_pull_repo(repo_name, repo_path) return True def update_search_index(): """Update the search index with all markdown files.""" try: writer = ix.writer() for repo_dir in Path(REPOS_DIR).iterdir(): if not repo_dir.is_dir(): continue for md_file in repo_dir.rglob('*'): if md_file.suffix.lower() in MARKDOWN_EXTENSIONS: try: content = md_file.read_text(encoding='utf-8', errors='ignore') # Extract title from filename or first heading title = md_file.stem rel_path = str(md_file.relative_to(REPOS_DIR)) writer.add_document( path=rel_path, title=title, content=content ) except Exception as e: logger.error(f"Error indexing {md_file}: {e}") writer.commit() logger.info("Search index updated successfully") except Exception as e: logger.error(f"Error updating search index: {e}") def scheduled_pull(): """Perform scheduled pulls of all repositories.""" logger.info("Starting scheduled repository pull") repos = load_repositories() for repo_name, repo_config in repos.items(): if not repo_config.get('enabled', True): continue clone_or_pull(repo_name, repo_config['url']) # Update search index after pulling update_search_index() logger.info("Scheduled pull completed") def setup_scheduler(): """Setup the background scheduler for periodic pulls.""" scheduler = BackgroundScheduler() repos = load_repositories() for repo_name, repo_config in repos.items(): if repo_config.get('enabled', True): cron_schedule = repo_config.get('schedule', '0 */6 * * *') # Default: every 6 hours try: scheduler.add_job( scheduled_pull, CronTrigger.from_crontab(cron_schedule), id=f"pull_{repo_name}", name=f"Pull {repo_name}" ) logger.info(f"Scheduled pull for {repo_name}: {cron_schedule}") except Exception as e: logger.error(f"Error scheduling {repo_name}: {e}") scheduler.start() return scheduler # Routes @app.route('/') def index(): """Display the main page with list of repositories and search.""" repos = load_repositories() repo_list = [] for repo_name in repos.keys(): repo_path = os.path.join(REPOS_DIR, repo_name) markdown_files = [] if os.path.exists(repo_path): for md_file in Path(repo_path).rglob('*'): if md_file.suffix.lower() in MARKDOWN_EXTENSIONS: rel_path = str(md_file.relative_to(repo_path)) markdown_files.append({ 'name': md_file.stem, 'path': rel_path }) repo_list.append({ 'name': repo_name, 'files': sorted(markdown_files, key=lambda x: x['path']) }) return render_template('index.html', repositories=repo_list) @app.route('/api/search', methods=['GET']) def search(): """Search markdown files.""" query_str = request.args.get('q', '').strip() if not query_str or len(query_str) < 2: return jsonify({'results': []}) try: with ix.searcher() as searcher: query = QueryParser("content", ix.schema).parse(query_str) results = searcher.search(query) search_results = [] for hit in results: search_results.append({ 'path': hit['path'], 'title': hit['title'] }) return jsonify({'results': search_results}) except Exception as e: logger.error(f"Search error: {e}") return jsonify({'results': [], 'error': str(e)}) @app.route('/view//') def view_file(repo, filepath): """View a markdown file converted to HTML.""" # Security: prevent directory traversal if '..' in filepath or filepath.startswith('/'): return "Invalid path", 400 file_path = os.path.join(REPOS_DIR, repo, filepath) # Ensure the file is within the repo directory try: file_path = os.path.realpath(file_path) repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo)) if not file_path.startswith(repo_path): return "Access denied", 403 except Exception: return "Invalid path", 400 if not os.path.exists(file_path): return "File not found", 404 try: with open(file_path, 'r', encoding='utf-8') as f: md_content = f.read() # Convert markdown to HTML html_content = markdown.markdown( md_content, extensions=['extra', 'codehilite', 'toc'] ) return render_template( 'view.html', repo=repo, filepath=filepath, content=html_content, filename=os.path.basename(filepath) ) except Exception as e: logger.error(f"Error reading file {file_path}: {e}") return f"Error reading file: {e}", 500 @app.route('/api/status') def status(): """Return application status.""" repos = load_repositories() repo_status = {} for repo_name in repos.keys(): repo_path = os.path.join(REPOS_DIR, repo_name) repo_status[repo_name] = { 'cloned': os.path.exists(repo_path), 'last_modified': datetime.fromtimestamp( os.path.getmtime(repo_path) ).isoformat() if os.path.exists(repo_path) else None } return jsonify({ 'status': 'running', 'repositories': repo_status, 'timestamp': datetime.now().isoformat() }) if __name__ == '__main__': # Perform initial pull and index logger.info("Initializing markmywords") scheduled_pull() # Setup scheduler scheduler = setup_scheduler() # Start Flask app app.run(host='0.0.0.0', port=5000, debug=False)