SHA256
405 lines
13 KiB
Python
405 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
markmywords - Self-hosted markdown repository viewer with search
|
|
"""
|
|
|
|
import os
|
|
import subprocess
|
|
import json
|
|
from pathlib import Path
|
|
from datetime import datetime
|
|
from flask import Flask, render_template, request, jsonify, send_file
|
|
from apscheduler.schedulers.background import BackgroundScheduler
|
|
from apscheduler.triggers.cron import CronTrigger
|
|
import markdown
|
|
import logging
|
|
import re
|
|
from whoosh.index import create_in, open_dir
|
|
from whoosh.fields import Schema, TEXT, ID
|
|
from whoosh.qparser import QueryParser
|
|
|
|
app = Flask(__name__)
|
|
|
|
# Configuration
|
|
CONFIG_FILE = "/config/repositories.json"
|
|
REPOS_DIR = "/data/repos"
|
|
INDEX_DIR = "/data/search_index"
|
|
MARKDOWN_EXTENSIONS = ['.md', '.markdown']
|
|
|
|
# Setup logging
|
|
logging.basicConfig(level=logging.INFO)
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Ensure directories exist
|
|
os.makedirs(REPOS_DIR, exist_ok=True)
|
|
os.makedirs(INDEX_DIR, exist_ok=True)
|
|
|
|
# Initialize search index
|
|
def create_search_index():
|
|
"""Create or open the search index."""
|
|
schema = Schema(
|
|
path=ID(stored=True),
|
|
title=TEXT(stored=True),
|
|
content=TEXT(stored=False)
|
|
)
|
|
|
|
if not os.path.exists(os.path.join(INDEX_DIR, "WRITELOCK")):
|
|
return create_in(INDEX_DIR, schema)
|
|
return open_dir(INDEX_DIR)
|
|
|
|
ix = create_search_index()
|
|
|
|
def is_hidden_path(file_path):
|
|
"""Check if a file path contains any hidden directories or files (starting with dot)."""
|
|
path_obj = Path(file_path) if isinstance(file_path, str) else file_path
|
|
# Check all parts of the path including the filename
|
|
return any(part.startswith('.') for part in path_obj.parts)
|
|
|
|
def rewrite_image_paths(md_content, repo, filepath):
|
|
"""Rewrite relative image paths to be served by Flask."""
|
|
# Get the directory of the current file (relative to repo root)
|
|
file_dir = os.path.dirname(filepath)
|
|
|
|
# Pattern to match markdown image syntax: 
|
|
def replace_image_path(match):
|
|
alt_text = match.group(1)
|
|
img_path = match.group(2)
|
|
|
|
# Skip absolute URLs and data URIs
|
|
if img_path.startswith(('http://', 'https://', 'data:')):
|
|
return match.group(0)
|
|
|
|
# Resolve relative paths
|
|
if img_path.startswith('/'):
|
|
# Absolute path from repo root
|
|
resolved_path = img_path.lstrip('/')
|
|
else:
|
|
# Relative path - resolve it relative to the file's directory
|
|
if file_dir:
|
|
resolved_path = os.path.normpath(os.path.join(file_dir, img_path))
|
|
else:
|
|
resolved_path = img_path
|
|
|
|
# Ensure forward slashes for URL
|
|
resolved_path = resolved_path.replace(os.sep, '/')
|
|
image_url = f"/image/{repo}/{resolved_path}"
|
|
|
|
return f""
|
|
|
|
# Replace all markdown image references
|
|
md_content = re.sub(r'!\[([^\]]*)\]\(([^\)]+)\)', replace_image_path, md_content)
|
|
return md_content
|
|
|
|
def build_directory_tree(repo_path):
|
|
"""Build a hierarchical tree of markdown files organized by directory."""
|
|
tree = {}
|
|
|
|
if not os.path.exists(repo_path):
|
|
logger.warning(f"Repo path does not exist: {repo_path}")
|
|
return tree
|
|
|
|
file_count = 0
|
|
for md_file in Path(repo_path).rglob('*'):
|
|
if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS:
|
|
continue
|
|
if is_hidden_path(md_file):
|
|
logger.debug(f"Skipping hidden path: {md_file}")
|
|
continue
|
|
|
|
file_count += 1
|
|
rel_path = md_file.relative_to(repo_path)
|
|
parts = rel_path.parts[:-1] # All parts except filename
|
|
filename = md_file.stem # Name without extension
|
|
|
|
logger.debug(f"Adding to tree: {rel_path} (name: {filename})")
|
|
|
|
# Navigate/create nested dict structure
|
|
current = tree
|
|
for part in parts:
|
|
if part not in current:
|
|
current[part] = {}
|
|
current = current[part]
|
|
|
|
# Add file to current level
|
|
if '_files' not in current:
|
|
current['_files'] = []
|
|
current['_files'].append({
|
|
'name': filename,
|
|
'path': str(rel_path)
|
|
})
|
|
|
|
logger.info(f"Built tree for {repo_path}: found {file_count} markdown files")
|
|
return tree
|
|
|
|
def load_repositories():
|
|
"""Load repository configuration from file."""
|
|
if not os.path.exists(CONFIG_FILE):
|
|
logger.warning(f"Config file not found: {CONFIG_FILE}")
|
|
return {}
|
|
|
|
try:
|
|
with open(CONFIG_FILE, 'r') as f:
|
|
return json.load(f)
|
|
except Exception as e:
|
|
logger.error(f"Error loading config: {e}")
|
|
return {}
|
|
|
|
def git_pull_repo(repo_name, repo_path):
|
|
"""Perform a git pull on the specified repository."""
|
|
try:
|
|
logger.info(f"Pulling repository: {repo_name}")
|
|
result = subprocess.run(
|
|
["git", "pull"],
|
|
cwd=repo_path,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300
|
|
)
|
|
logger.info(f"Pull result for {repo_name}: {result.stdout}")
|
|
if result.returncode != 0:
|
|
logger.error(f"Pull error for {repo_name}: {result.stderr}")
|
|
return result.returncode == 0
|
|
except subprocess.TimeoutExpired:
|
|
logger.error(f"Git pull timeout for {repo_name}")
|
|
return False
|
|
except Exception as e:
|
|
logger.error(f"Error pulling {repo_name}: {e}")
|
|
return False
|
|
|
|
def clone_or_pull(repo_name, repo_url):
|
|
"""Clone repository if it doesn't exist, otherwise pull."""
|
|
repo_path = os.path.join(REPOS_DIR, repo_name)
|
|
|
|
if not os.path.exists(repo_path):
|
|
try:
|
|
logger.info(f"Cloning repository: {repo_name} from {repo_url}")
|
|
subprocess.run(
|
|
["git", "clone", repo_url, repo_path],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300
|
|
)
|
|
logger.info(f"Successfully cloned {repo_name}")
|
|
except Exception as e:
|
|
logger.error(f"Error cloning {repo_name}: {e}")
|
|
return False
|
|
else:
|
|
return git_pull_repo(repo_name, repo_path)
|
|
|
|
return True
|
|
|
|
def update_search_index():
|
|
"""Update the search index with all markdown files."""
|
|
try:
|
|
writer = ix.writer()
|
|
|
|
for repo_dir in Path(REPOS_DIR).iterdir():
|
|
if not repo_dir.is_dir():
|
|
continue
|
|
|
|
for md_file in repo_dir.rglob('*'):
|
|
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS and not is_hidden_path(md_file):
|
|
try:
|
|
content = md_file.read_text(encoding='utf-8', errors='ignore')
|
|
# Extract title from filename or first heading
|
|
title = md_file.stem
|
|
rel_path = str(md_file.relative_to(REPOS_DIR))
|
|
|
|
writer.add_document(
|
|
path=rel_path,
|
|
title=title,
|
|
content=content
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Error indexing {md_file}: {e}")
|
|
|
|
writer.commit()
|
|
logger.info("Search index updated successfully")
|
|
except Exception as e:
|
|
logger.error(f"Error updating search index: {e}")
|
|
|
|
def scheduled_pull():
|
|
"""Perform scheduled pulls of all repositories."""
|
|
logger.info("Starting scheduled repository pull")
|
|
repos = load_repositories()
|
|
|
|
for repo_name, repo_config in repos.items():
|
|
if not repo_config.get('enabled', True):
|
|
continue
|
|
|
|
clone_or_pull(repo_name, repo_config['url'])
|
|
|
|
# Update search index after pulling
|
|
update_search_index()
|
|
logger.info("Scheduled pull completed")
|
|
|
|
def setup_scheduler():
|
|
"""Setup the background scheduler for periodic pulls."""
|
|
scheduler = BackgroundScheduler()
|
|
repos = load_repositories()
|
|
|
|
for repo_name, repo_config in repos.items():
|
|
if repo_config.get('enabled', True):
|
|
cron_schedule = repo_config.get('schedule', '0 */6 * * *') # Default: every 6 hours
|
|
try:
|
|
scheduler.add_job(
|
|
scheduled_pull,
|
|
CronTrigger.from_crontab(cron_schedule),
|
|
id=f"pull_{repo_name}",
|
|
name=f"Pull {repo_name}"
|
|
)
|
|
logger.info(f"Scheduled pull for {repo_name}: {cron_schedule}")
|
|
except Exception as e:
|
|
logger.error(f"Error scheduling {repo_name}: {e}")
|
|
|
|
scheduler.start()
|
|
return scheduler
|
|
|
|
# Routes
|
|
@app.route('/')
|
|
def index():
|
|
"""Display the main page with list of repositories and search."""
|
|
repos = load_repositories()
|
|
repo_list = []
|
|
|
|
for repo_name in repos.keys():
|
|
repo_path = os.path.join(REPOS_DIR, repo_name)
|
|
file_tree = build_directory_tree(repo_path)
|
|
|
|
repo_list.append({
|
|
'name': repo_name,
|
|
'tree': file_tree
|
|
})
|
|
|
|
return render_template('index.html', repositories=repo_list)
|
|
|
|
@app.route('/image/<repo>/<path:filepath>')
|
|
def serve_image(repo, filepath):
|
|
"""Serve images from repository directories."""
|
|
# Security: prevent directory traversal
|
|
if '..' in filepath or filepath.startswith('/'):
|
|
return "Invalid path", 400
|
|
|
|
file_path = os.path.join(REPOS_DIR, repo, filepath)
|
|
|
|
# Ensure the file is within the repo directory
|
|
try:
|
|
file_path = os.path.realpath(file_path)
|
|
repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo))
|
|
if not file_path.startswith(repo_path):
|
|
return "Access denied", 403
|
|
except Exception:
|
|
return "Invalid path", 400
|
|
|
|
if not os.path.exists(file_path):
|
|
logger.warning(f"Image not found: {file_path}")
|
|
return "Image not found", 404
|
|
|
|
try:
|
|
return send_file(file_path)
|
|
except Exception as e:
|
|
logger.error(f"Error serving image {file_path}: {e}")
|
|
return f"Error serving image: {e}", 500
|
|
|
|
@app.route('/api/search', methods=['GET'])
|
|
def search():
|
|
"""Search markdown files."""
|
|
query_str = request.args.get('q', '').strip()
|
|
|
|
if not query_str or len(query_str) < 2:
|
|
return jsonify({'results': []})
|
|
|
|
try:
|
|
with ix.searcher() as searcher:
|
|
query = QueryParser("content", ix.schema).parse(query_str)
|
|
results = searcher.search(query)
|
|
|
|
search_results = []
|
|
for hit in results:
|
|
search_results.append({
|
|
'path': hit['path'],
|
|
'title': hit['title']
|
|
})
|
|
|
|
return jsonify({'results': search_results})
|
|
except Exception as e:
|
|
logger.error(f"Search error: {e}")
|
|
return jsonify({'results': [], 'error': str(e)})
|
|
|
|
@app.route('/view/<repo>/<path:filepath>')
|
|
def view_file(repo, filepath):
|
|
"""View a markdown file converted to HTML."""
|
|
# Security: prevent directory traversal
|
|
if '..' in filepath or filepath.startswith('/'):
|
|
return "Invalid path", 400
|
|
|
|
file_path = os.path.join(REPOS_DIR, repo, filepath)
|
|
|
|
# Ensure the file is within the repo directory
|
|
try:
|
|
file_path = os.path.realpath(file_path)
|
|
repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo))
|
|
if not file_path.startswith(repo_path):
|
|
return "Access denied", 403
|
|
except Exception:
|
|
return "Invalid path", 400
|
|
|
|
if not os.path.exists(file_path):
|
|
return "File not found", 404
|
|
|
|
try:
|
|
with open(file_path, 'r', encoding='utf-8') as f:
|
|
md_content = f.read()
|
|
|
|
# Rewrite image paths to be served by Flask
|
|
md_content = rewrite_image_paths(md_content, repo, filepath)
|
|
|
|
# Convert markdown to HTML
|
|
html_content = markdown.markdown(
|
|
md_content,
|
|
extensions=['extra', 'codehilite', 'toc']
|
|
)
|
|
|
|
return render_template(
|
|
'view.html',
|
|
repo=repo,
|
|
filepath=filepath,
|
|
content=html_content,
|
|
filename=os.path.basename(filepath)
|
|
)
|
|
except Exception as e:
|
|
logger.error(f"Error reading file {file_path}: {e}")
|
|
return f"Error reading file: {e}", 500
|
|
|
|
@app.route('/api/status')
|
|
def status():
|
|
"""Return application status."""
|
|
repos = load_repositories()
|
|
repo_status = {}
|
|
|
|
for repo_name in repos.keys():
|
|
repo_path = os.path.join(REPOS_DIR, repo_name)
|
|
repo_status[repo_name] = {
|
|
'cloned': os.path.exists(repo_path),
|
|
'last_modified': datetime.fromtimestamp(
|
|
os.path.getmtime(repo_path)
|
|
).isoformat() if os.path.exists(repo_path) else None
|
|
}
|
|
|
|
return jsonify({
|
|
'status': 'running',
|
|
'repositories': repo_status,
|
|
'timestamp': datetime.now().isoformat()
|
|
})
|
|
|
|
if __name__ == '__main__':
|
|
# Perform initial pull and index
|
|
logger.info("Initializing markmywords")
|
|
scheduled_pull()
|
|
|
|
# Setup scheduler
|
|
scheduler = setup_scheduler()
|
|
|
|
# Start Flask app
|
|
app.run(host='0.0.0.0', port=5000, debug=False)
|