SHA256
300 lines
9.4 KiB
Python
300 lines
9.4 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
markmywords - Self-hosted markdown repository viewer with search
|
|
"""
|
|
|
|
import os
|
|
import subprocess
|
|
import json
|
|
from pathlib import Path
|
|
from datetime import datetime
|
|
from flask import Flask, render_template, request, jsonify
|
|
from apscheduler.schedulers.background import BackgroundScheduler
|
|
from apscheduler.triggers.cron import CronTrigger
|
|
import markdown
|
|
import logging
|
|
from whoosh.index import create_in, open_dir
|
|
from whoosh.fields import Schema, TEXT, ID
|
|
from whoosh.qparser import QueryParser
|
|
|
|
app = Flask(__name__)
|
|
|
|
# Configuration
|
|
CONFIG_FILE = "/config/repositories.json"
|
|
REPOS_DIR = "/data/repos"
|
|
INDEX_DIR = "/data/search_index"
|
|
MARKDOWN_EXTENSIONS = ['.md', '.markdown']
|
|
|
|
# Setup logging
|
|
logging.basicConfig(level=logging.INFO)
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Ensure directories exist
|
|
os.makedirs(REPOS_DIR, exist_ok=True)
|
|
os.makedirs(INDEX_DIR, exist_ok=True)
|
|
|
|
# Initialize search index
|
|
def create_search_index():
|
|
"""Create or open the search index."""
|
|
schema = Schema(
|
|
path=ID(stored=True),
|
|
title=TEXT(stored=True),
|
|
content=TEXT(stored=False)
|
|
)
|
|
|
|
if not os.path.exists(os.path.join(INDEX_DIR, "WRITELOCK")):
|
|
return create_in(INDEX_DIR, schema)
|
|
return open_dir(INDEX_DIR)
|
|
|
|
ix = create_search_index()
|
|
|
|
def load_repositories():
|
|
"""Load repository configuration from file."""
|
|
if not os.path.exists(CONFIG_FILE):
|
|
logger.warning(f"Config file not found: {CONFIG_FILE}")
|
|
return {}
|
|
|
|
try:
|
|
with open(CONFIG_FILE, 'r') as f:
|
|
return json.load(f)
|
|
except Exception as e:
|
|
logger.error(f"Error loading config: {e}")
|
|
return {}
|
|
|
|
def git_pull_repo(repo_name, repo_path):
|
|
"""Perform a git pull on the specified repository."""
|
|
try:
|
|
logger.info(f"Pulling repository: {repo_name}")
|
|
result = subprocess.run(
|
|
["git", "pull"],
|
|
cwd=repo_path,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300
|
|
)
|
|
logger.info(f"Pull result for {repo_name}: {result.stdout}")
|
|
if result.returncode != 0:
|
|
logger.error(f"Pull error for {repo_name}: {result.stderr}")
|
|
return result.returncode == 0
|
|
except subprocess.TimeoutExpired:
|
|
logger.error(f"Git pull timeout for {repo_name}")
|
|
return False
|
|
except Exception as e:
|
|
logger.error(f"Error pulling {repo_name}: {e}")
|
|
return False
|
|
|
|
def clone_or_pull(repo_name, repo_url):
|
|
"""Clone repository if it doesn't exist, otherwise pull."""
|
|
repo_path = os.path.join(REPOS_DIR, repo_name)
|
|
|
|
if not os.path.exists(repo_path):
|
|
try:
|
|
logger.info(f"Cloning repository: {repo_name} from {repo_url}")
|
|
subprocess.run(
|
|
["git", "clone", repo_url, repo_path],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300
|
|
)
|
|
logger.info(f"Successfully cloned {repo_name}")
|
|
except Exception as e:
|
|
logger.error(f"Error cloning {repo_name}: {e}")
|
|
return False
|
|
else:
|
|
return git_pull_repo(repo_name, repo_path)
|
|
|
|
return True
|
|
|
|
def update_search_index():
|
|
"""Update the search index with all markdown files."""
|
|
try:
|
|
writer = ix.writer()
|
|
|
|
for repo_dir in Path(REPOS_DIR).iterdir():
|
|
if not repo_dir.is_dir():
|
|
continue
|
|
|
|
for md_file in repo_dir.rglob('*'):
|
|
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS:
|
|
try:
|
|
content = md_file.read_text(encoding='utf-8', errors='ignore')
|
|
# Extract title from filename or first heading
|
|
title = md_file.stem
|
|
rel_path = str(md_file.relative_to(REPOS_DIR))
|
|
|
|
writer.add_document(
|
|
path=rel_path,
|
|
title=title,
|
|
content=content
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Error indexing {md_file}: {e}")
|
|
|
|
writer.commit()
|
|
logger.info("Search index updated successfully")
|
|
except Exception as e:
|
|
logger.error(f"Error updating search index: {e}")
|
|
|
|
def scheduled_pull():
|
|
"""Perform scheduled pulls of all repositories."""
|
|
logger.info("Starting scheduled repository pull")
|
|
repos = load_repositories()
|
|
|
|
for repo_name, repo_config in repos.items():
|
|
if not repo_config.get('enabled', True):
|
|
continue
|
|
|
|
clone_or_pull(repo_name, repo_config['url'])
|
|
|
|
# Update search index after pulling
|
|
update_search_index()
|
|
logger.info("Scheduled pull completed")
|
|
|
|
def setup_scheduler():
|
|
"""Setup the background scheduler for periodic pulls."""
|
|
scheduler = BackgroundScheduler()
|
|
repos = load_repositories()
|
|
|
|
for repo_name, repo_config in repos.items():
|
|
if repo_config.get('enabled', True):
|
|
cron_schedule = repo_config.get('schedule', '0 */6 * * *') # Default: every 6 hours
|
|
try:
|
|
scheduler.add_job(
|
|
scheduled_pull,
|
|
CronTrigger.from_crontab(cron_schedule),
|
|
id=f"pull_{repo_name}",
|
|
name=f"Pull {repo_name}"
|
|
)
|
|
logger.info(f"Scheduled pull for {repo_name}: {cron_schedule}")
|
|
except Exception as e:
|
|
logger.error(f"Error scheduling {repo_name}: {e}")
|
|
|
|
scheduler.start()
|
|
return scheduler
|
|
|
|
# Routes
|
|
@app.route('/')
|
|
def index():
|
|
"""Display the main page with list of repositories and search."""
|
|
repos = load_repositories()
|
|
repo_list = []
|
|
|
|
for repo_name in repos.keys():
|
|
repo_path = os.path.join(REPOS_DIR, repo_name)
|
|
markdown_files = []
|
|
|
|
if os.path.exists(repo_path):
|
|
for md_file in Path(repo_path).rglob('*'):
|
|
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS:
|
|
rel_path = str(md_file.relative_to(repo_path))
|
|
markdown_files.append({
|
|
'name': md_file.stem,
|
|
'path': rel_path
|
|
})
|
|
|
|
repo_list.append({
|
|
'name': repo_name,
|
|
'files': sorted(markdown_files, key=lambda x: x['path'])
|
|
})
|
|
|
|
return render_template('index.html', repositories=repo_list)
|
|
|
|
@app.route('/api/search', methods=['GET'])
|
|
def search():
|
|
"""Search markdown files."""
|
|
query_str = request.args.get('q', '').strip()
|
|
|
|
if not query_str or len(query_str) < 2:
|
|
return jsonify({'results': []})
|
|
|
|
try:
|
|
with ix.searcher() as searcher:
|
|
query = QueryParser("content", ix.schema).parse(query_str)
|
|
results = searcher.search(query)
|
|
|
|
search_results = []
|
|
for hit in results:
|
|
search_results.append({
|
|
'path': hit['path'],
|
|
'title': hit['title']
|
|
})
|
|
|
|
return jsonify({'results': search_results})
|
|
except Exception as e:
|
|
logger.error(f"Search error: {e}")
|
|
return jsonify({'results': [], 'error': str(e)})
|
|
|
|
@app.route('/view/<repo>/<path:filepath>')
|
|
def view_file(repo, filepath):
|
|
"""View a markdown file converted to HTML."""
|
|
# Security: prevent directory traversal
|
|
if '..' in filepath or filepath.startswith('/'):
|
|
return "Invalid path", 400
|
|
|
|
file_path = os.path.join(REPOS_DIR, repo, filepath)
|
|
|
|
# Ensure the file is within the repo directory
|
|
try:
|
|
file_path = os.path.realpath(file_path)
|
|
repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo))
|
|
if not file_path.startswith(repo_path):
|
|
return "Access denied", 403
|
|
except Exception:
|
|
return "Invalid path", 400
|
|
|
|
if not os.path.exists(file_path):
|
|
return "File not found", 404
|
|
|
|
try:
|
|
with open(file_path, 'r', encoding='utf-8') as f:
|
|
md_content = f.read()
|
|
|
|
# Convert markdown to HTML
|
|
html_content = markdown.markdown(
|
|
md_content,
|
|
extensions=['extra', 'codehilite', 'toc']
|
|
)
|
|
|
|
return render_template(
|
|
'view.html',
|
|
repo=repo,
|
|
filepath=filepath,
|
|
content=html_content,
|
|
filename=os.path.basename(filepath)
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Error reading file {file_path}: {e}")
|
|
return f"Error reading file: {e}", 500
|
|
|
|
@app.route('/api/status')
|
|
def status():
|
|
"""Return application status."""
|
|
repos = load_repositories()
|
|
repo_status = {}
|
|
|
|
for repo_name in repos.keys():
|
|
repo_path = os.path.join(REPOS_DIR, repo_name)
|
|
repo_status[repo_name] = {
|
|
'cloned': os.path.exists(repo_path),
|
|
'last_modified': datetime.fromtimestamp(
|
|
os.path.getmtime(repo_path)
|
|
).isoformat() if os.path.exists(repo_path) else None
|
|
}
|
|
|
|
return jsonify({
|
|
'status': 'running',
|
|
'repositories': repo_status,
|
|
'timestamp': datetime.now().isoformat()
|
|
})
|
|
|
|
if __name__ == '__main__':
|
|
# Perform initial pull and index
|
|
logger.info("Initializing markmywords")
|
|
scheduled_pull()
|
|
|
|
# Setup scheduler
|
|
scheduler = setup_scheduler()
|
|
|
|
# Start Flask app
|
|
app.run(host='0.0.0.0', port=5000, debug=False)
|