Updated to do local content w/ passwording and encryption

This commit is contained in:
Discsearcher
2026-08-28 14:59:19 -04:00
parent 2dc05e72d4
commit af59158ff9
9 changed files with 1642 additions and 233 deletions
+321 -181
View File
@@ -1,38 +1,125 @@
#!/usr/bin/env python3
"""
markmywords - Self-hosted markdown repository viewer with search
markmywords - Self-hosted markdown viewer with search and encryption
"""
import os
import subprocess
import json
from pathlib import Path
from datetime import datetime
from flask import Flask, render_template, request, jsonify, send_file
from apscheduler.schedulers.background import BackgroundScheduler
from apscheduler.triggers.cron import CronTrigger
from flask import Flask, render_template, request, jsonify, send_file, session, redirect, url_for
from werkzeug.security import generate_password_hash, check_password_hash
import markdown
import logging
import re
from whoosh.index import create_in, open_dir
from whoosh.fields import Schema, TEXT, ID
from whoosh.qparser import QueryParser
from cryptography.fernet import Fernet
from cryptography.hazmat.primitives import hashes
from cryptography.hazmat.primitives.kdf.pbkdf2 import PBKDF2HMAC
from cryptography.hazmat.backends import default_backend
import base64
app = Flask(__name__)
app.secret_key = os.environ.get('FLASK_SECRET_KEY', 'dev-secret-key-change-in-production')
# Configuration
CONFIG_FILE = "/config/repositories.json"
REPOS_DIR = "/data/repos"
CONTENT_DIR = "/content"
INDEX_DIR = "/data/search_index"
PASSWORD_FILE = "/config/page_passwords.json"
MARKDOWN_EXTENSIONS = ['.md', '.markdown']
ENCRYPTION_ENABLED = os.environ.get('ENCRYPTION_ENABLED', 'true').lower() == 'true'
# Setup logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# Ensure directories exist
os.makedirs(REPOS_DIR, exist_ok=True)
os.makedirs(CONTENT_DIR, exist_ok=True)
os.makedirs(INDEX_DIR, exist_ok=True)
os.makedirs(os.path.dirname(PASSWORD_FILE), exist_ok=True)
def load_passwords():
"""Load password database from file."""
if not os.path.exists(PASSWORD_FILE):
return {}
try:
with open(PASSWORD_FILE, 'r') as f:
return json.load(f)
except Exception as e:
logger.error(f"Error loading passwords: {e}")
return {}
def save_passwords(passwords):
"""Save password database to file."""
try:
os.makedirs(os.path.dirname(PASSWORD_FILE), exist_ok=True)
with open(PASSWORD_FILE, 'w') as f:
json.dump(passwords, f, indent=2)
os.chmod(PASSWORD_FILE, 0o600) # Restrict file permissions
return True
except Exception as e:
logger.error(f"Error saving passwords: {e}")
return False
def derive_encryption_key(password):
"""Derive an encryption key from a password."""
salt = b'markmywords_salt' # Fixed salt for consistency
kdf = PBKDF2HMAC(
algorithm=hashes.SHA256(),
length=32,
salt=salt,
iterations=100000,
backend=default_backend()
)
key = base64.urlsafe_b64encode(kdf.derive(password.encode()))
return key
def encrypt_file_content(content, password):
"""Encrypt file content using password-derived key."""
try:
key = derive_encryption_key(password)
f = Fernet(key)
encrypted = f.encrypt(content.encode('utf-8'))
return encrypted.decode('utf-8')
except Exception as e:
logger.error(f"Error encrypting content: {e}")
return None
def decrypt_file_content(encrypted_content, password):
"""Decrypt file content using password-derived key."""
try:
key = derive_encryption_key(password)
f = Fernet(key)
decrypted = f.decrypt(encrypted_content.encode('utf-8'))
return decrypted.decode('utf-8')
except Exception as e:
logger.error(f"Error decrypting content: {e}")
return None
def is_encrypted_file(filepath):
"""Check if a file is encrypted (has .enc extension)."""
return filepath.endswith('.enc')
def get_file_key(filepath):
"""Generate a consistent key for a file."""
return filepath
def is_file_protected(filepath):
"""Check if a file is password protected."""
passwords = load_passwords()
file_key = get_file_key(filepath)
return passwords.get(file_key, {}).get('protected', False)
def is_authenticated(filepath):
"""Check if user is authenticated for a protected file."""
if not is_file_protected(filepath):
return True # Not protected, so allowed
file_key = get_file_key(filepath)
authenticated_files = session.get('authenticated_files', {})
return authenticated_files.get(file_key, False)
# Initialize search index
def create_search_index():
@@ -55,9 +142,9 @@ def is_hidden_path(file_path):
# Check all parts of the path including the filename
return any(part.startswith('.') for part in path_obj.parts)
def rewrite_image_paths(md_content, repo, filepath):
def rewrite_image_paths(md_content, filepath):
"""Rewrite relative image paths to be served by Flask."""
# Get the directory of the current file (relative to repo root)
# Get the directory of the current file
file_dir = os.path.dirname(filepath)
# Pattern to match markdown image syntax: ![alt](path)
@@ -71,7 +158,7 @@ def rewrite_image_paths(md_content, repo, filepath):
# Resolve relative paths
if img_path.startswith('/'):
# Absolute path from repo root
# Absolute path from content root
resolved_path = img_path.lstrip('/')
else:
# Relative path - resolve it relative to the file's directory
@@ -82,7 +169,7 @@ def rewrite_image_paths(md_content, repo, filepath):
# Ensure forward slashes for URL
resolved_path = resolved_path.replace(os.sep, '/')
image_url = f"/image/{repo}/{resolved_path}"
image_url = f"/image/{resolved_path}"
return f"![{alt_text}]({image_url})"
@@ -90,26 +177,37 @@ def rewrite_image_paths(md_content, repo, filepath):
md_content = re.sub(r'!\[([^\]]*)\]\(([^\)]+)\)', replace_image_path, md_content)
return md_content
def build_directory_tree(repo_path):
def build_directory_tree():
"""Build a hierarchical tree of markdown files organized by directory."""
tree = {}
if not os.path.exists(repo_path):
logger.warning(f"Repo path does not exist: {repo_path}")
if not os.path.exists(CONTENT_DIR):
logger.warning(f"Content directory does not exist: {CONTENT_DIR}")
return tree
file_count = 0
for md_file in Path(repo_path).rglob('*'):
if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS:
# Include both .md and .md.enc files
for md_file in Path(CONTENT_DIR).rglob('*'):
is_md = md_file.suffix.lower() in MARKDOWN_EXTENSIONS
is_enc = md_file.suffix == '.enc' and md_file.stem.endswith('.md')
if not (is_md or is_enc):
continue
if is_hidden_path(md_file):
logger.debug(f"Skipping hidden path: {md_file}")
continue
file_count += 1
rel_path = md_file.relative_to(repo_path)
rel_path = md_file.relative_to(CONTENT_DIR)
parts = rel_path.parts[:-1] # All parts except filename
filename = md_file.stem # Name without extension
# For encrypted files, show without .enc extension
if is_enc:
filename = md_file.stem # Removes .enc, keeping the .md
if filename.endswith('.md'):
filename = filename[:-3] # Remove .md to show clean name
else:
filename = md_file.stem # Name without extension
logger.debug(f"Adding to tree: {rel_path} (name: {filename})")
@@ -125,168 +223,73 @@ def build_directory_tree(repo_path):
current['_files'] = []
current['_files'].append({
'name': filename,
'path': str(rel_path)
'path': str(rel_path),
'encrypted': is_enc
})
logger.info(f"Built tree for {repo_path}: found {file_count} markdown files")
logger.info(f"Built tree for {CONTENT_DIR}: found {file_count} markdown files")
return tree
def load_repositories():
"""Load repository configuration from file."""
if not os.path.exists(CONFIG_FILE):
logger.warning(f"Config file not found: {CONFIG_FILE}")
return {}
try:
with open(CONFIG_FILE, 'r') as f:
return json.load(f)
except Exception as e:
logger.error(f"Error loading config: {e}")
return {}
def git_pull_repo(repo_name, repo_path):
"""Perform a git pull on the specified repository."""
try:
logger.info(f"Pulling repository: {repo_name}")
result = subprocess.run(
["git", "pull"],
cwd=repo_path,
capture_output=True,
text=True,
timeout=300
)
logger.info(f"Pull result for {repo_name}: {result.stdout}")
if result.returncode != 0:
logger.error(f"Pull error for {repo_name}: {result.stderr}")
return result.returncode == 0
except subprocess.TimeoutExpired:
logger.error(f"Git pull timeout for {repo_name}")
return False
except Exception as e:
logger.error(f"Error pulling {repo_name}: {e}")
return False
def clone_or_pull(repo_name, repo_url):
"""Clone repository if it doesn't exist, otherwise pull."""
repo_path = os.path.join(REPOS_DIR, repo_name)
if not os.path.exists(repo_path):
try:
logger.info(f"Cloning repository: {repo_name} from {repo_url}")
subprocess.run(
["git", "clone", repo_url, repo_path],
capture_output=True,
text=True,
timeout=300
)
logger.info(f"Successfully cloned {repo_name}")
except Exception as e:
logger.error(f"Error cloning {repo_name}: {e}")
return False
else:
return git_pull_repo(repo_name, repo_path)
return True
def update_search_index():
"""Update the search index with all markdown files."""
try:
writer = ix.writer()
for repo_dir in Path(REPOS_DIR).iterdir():
if not repo_dir.is_dir():
for md_file in Path(CONTENT_DIR).rglob('*'):
if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS and not md_file.suffix == '.enc':
continue
if is_hidden_path(md_file):
continue
for md_file in repo_dir.rglob('*'):
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS and not is_hidden_path(md_file):
try:
content = md_file.read_text(encoding='utf-8', errors='ignore')
# Extract title from filename or first heading
title = md_file.stem
rel_path = str(md_file.relative_to(REPOS_DIR))
writer.add_document(
path=rel_path,
title=title,
content=content
)
except Exception as e:
logger.error(f"Error indexing {md_file}: {e}")
try:
# Handle encrypted files
rel_path = str(md_file.relative_to(CONTENT_DIR))
title = md_file.stem
if is_encrypted_file(rel_path):
# Skip encrypted files in search index (they need authentication)
logger.debug(f"Skipping encrypted file from search index: {rel_path}")
continue
content = md_file.read_text(encoding='utf-8', errors='ignore')
writer.add_document(
path=rel_path,
title=title,
content=content
)
except Exception as e:
logger.error(f"Error indexing {md_file}: {e}")
writer.commit()
logger.info("Search index updated successfully")
except Exception as e:
logger.error(f"Error updating search index: {e}")
def scheduled_pull():
"""Perform scheduled pulls of all repositories."""
logger.info("Starting scheduled repository pull")
repos = load_repositories()
for repo_name, repo_config in repos.items():
if not repo_config.get('enabled', True):
continue
clone_or_pull(repo_name, repo_config['url'])
# Update search index after pulling
update_search_index()
logger.info("Scheduled pull completed")
def setup_scheduler():
"""Setup the background scheduler for periodic pulls."""
scheduler = BackgroundScheduler()
repos = load_repositories()
for repo_name, repo_config in repos.items():
if repo_config.get('enabled', True):
cron_schedule = repo_config.get('schedule', '0 */6 * * *') # Default: every 6 hours
try:
scheduler.add_job(
scheduled_pull,
CronTrigger.from_crontab(cron_schedule),
id=f"pull_{repo_name}",
name=f"Pull {repo_name}"
)
logger.info(f"Scheduled pull for {repo_name}: {cron_schedule}")
except Exception as e:
logger.error(f"Error scheduling {repo_name}: {e}")
scheduler.start()
return scheduler
# Routes
@app.route('/')
def index():
"""Display the main page with list of repositories and search."""
repos = load_repositories()
repo_list = []
"""Display the main page with file navigation and search."""
file_tree = build_directory_tree()
for repo_name in repos.keys():
repo_path = os.path.join(REPOS_DIR, repo_name)
file_tree = build_directory_tree(repo_path)
repo_list.append({
'name': repo_name,
'tree': file_tree
})
return render_template('index.html', repositories=repo_list)
return render_template('index.html', repositories=[{
'name': 'Content',
'tree': file_tree
}])
@app.route('/image/<repo>/<path:filepath>')
def serve_image(repo, filepath):
"""Serve images from repository directories."""
@app.route('/image/<path:filepath>')
def serve_image(filepath):
"""Serve images from content directory."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return "Invalid path", 400
file_path = os.path.join(REPOS_DIR, repo, filepath)
file_path = os.path.join(CONTENT_DIR, filepath)
# Ensure the file is within the repo directory
# Ensure the file is within the content directory
try:
file_path = os.path.realpath(file_path)
repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo))
if not file_path.startswith(repo_path):
content_path = os.path.realpath(CONTENT_DIR)
if not file_path.startswith(content_path):
return "Access denied", 403
except Exception:
return "Invalid path", 400
@@ -326,20 +329,33 @@ def search():
logger.error(f"Search error: {e}")
return jsonify({'results': [], 'error': str(e)})
@app.route('/view/<repo>/<path:filepath>')
def view_file(repo, filepath):
@app.route('/view/<path:filepath>')
def view_file(filepath):
"""View a markdown file converted to HTML."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return "Invalid path", 400
file_path = os.path.join(REPOS_DIR, repo, filepath)
# Determine if looking for encrypted version
enc_filepath = filepath + '.enc' if not filepath.endswith('.enc') else filepath
# Ensure the file is within the repo directory
# Try encrypted file first if it exists
file_path = os.path.join(CONTENT_DIR, enc_filepath)
is_encrypted = False
if os.path.exists(file_path):
is_encrypted = True
lookup_filepath = enc_filepath
else:
# Fall back to regular file
file_path = os.path.join(CONTENT_DIR, filepath)
lookup_filepath = filepath
# Ensure the file is within the content directory
try:
file_path = os.path.realpath(file_path)
repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo))
if not file_path.startswith(repo_path):
content_path = os.path.realpath(CONTENT_DIR)
if not file_path.startswith(content_path):
return "Access denied", 403
except Exception:
return "Invalid path", 400
@@ -347,12 +363,38 @@ def view_file(repo, filepath):
if not os.path.exists(file_path):
return "File not found", 404
# Check if file is protected and user is authenticated
if is_file_protected(lookup_filepath):
if not is_authenticated(lookup_filepath):
return render_template(
'password_prompt.html',
filepath=lookup_filepath,
filename=os.path.basename(filepath)
)
try:
with open(file_path, 'r', encoding='utf-8') as f:
md_content = f.read()
# Read file content
if is_encrypted:
with open(file_path, 'r', encoding='utf-8') as f:
encrypted_content = f.read()
# Get password from session
file_key = get_file_key(lookup_filepath)
authenticated_files = session.get('authenticated_files', {})
password = authenticated_files.get(file_key + '_password')
if not password:
return "Unable to decrypt: password not found in session", 500
md_content = decrypt_file_content(encrypted_content, password)
if md_content is None:
return "Failed to decrypt file", 500
else:
with open(file_path, 'r', encoding='utf-8') as f:
md_content = f.read()
# Rewrite image paths to be served by Flask
md_content = rewrite_image_paths(md_content, repo, filepath)
md_content = rewrite_image_paths(md_content, filepath)
# Convert markdown to HTML
html_content = markdown.markdown(
@@ -362,43 +404,141 @@ def view_file(repo, filepath):
return render_template(
'view.html',
repo=repo,
filepath=filepath,
filepath=lookup_filepath,
content=html_content,
filename=os.path.basename(filepath)
filename=os.path.basename(filepath),
is_protected=is_file_protected(lookup_filepath),
is_encrypted=is_encrypted
)
except Exception as e:
logger.error(f"Error reading file {file_path}: {e}")
return f"Error reading file: {e}", 500
@app.route('/api/auth/<path:filepath>', methods=['POST'])
def authenticate(filepath):
"""Authenticate user for a protected file."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return jsonify({'success': False, 'error': 'Invalid path'}), 400
password = request.form.get('password', '')
file_key = get_file_key(filepath)
passwords = load_passwords()
file_data = passwords.get(file_key)
if not file_data or not file_data.get('protected'):
return jsonify({'success': False, 'error': 'File not protected'}), 400
# Check password
if check_password_hash(file_data['password_hash'], password):
# Store in session
if 'authenticated_files' not in session:
session['authenticated_files'] = {}
session['authenticated_files'][file_key] = True
# Store the password for decryption if file is encrypted
if is_encrypted_file(filepath):
session['authenticated_files'][file_key + '_password'] = password
session.modified = True
logger.info(f"User authenticated for {file_key}")
return jsonify({'success': True, 'redirect': url_for('view_file', filepath=filepath)})
else:
logger.warning(f"Failed authentication attempt for {file_key}")
return jsonify({'success': False, 'error': 'Invalid password'}), 401
@app.route('/api/protect/<path:filepath>', methods=['POST'])
def protect_file(filepath):
"""Protect or unprotect a file with a password."""
# This should be restricted to admin users in production
# For now, requires a master password via environment variable
master_password = os.environ.get('MARKMYWORDS_ADMIN_PASSWORD')
auth_header = request.headers.get('Authorization', '')
if not auth_header.startswith('Bearer '):
return jsonify({'success': False, 'error': 'Missing authorization'}), 401
token = auth_header.split(' ')[1]
if not master_password or token != master_password:
return jsonify({'success': False, 'error': 'Invalid authorization'}), 401
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return jsonify({'success': False, 'error': 'Invalid path'}), 400
action = request.json.get('action') # 'protect' or 'unprotect'
new_password = request.json.get('password')
encrypt_file = request.json.get('encrypt_file', False) # Whether to encrypt the file
file_key = get_file_key(filepath)
passwords = load_passwords()
if action == 'protect':
if not new_password:
return jsonify({'success': False, 'error': 'Password required'}), 400
passwords[file_key] = {
'protected': True,
'password_hash': generate_password_hash(new_password),
'created_at': datetime.now().isoformat(),
'encrypted': encrypt_file
}
# If encryption is requested, encrypt the file
if encrypt_file and ENCRYPTION_ENABLED:
try:
file_path = os.path.join(CONTENT_DIR, filepath)
if os.path.exists(file_path):
with open(file_path, 'r', encoding='utf-8') as f:
original_content = f.read()
encrypted_content = encrypt_file_content(original_content, new_password)
if encrypted_content:
# Save encrypted file with .enc extension
enc_file_path = file_path + '.enc'
with open(enc_file_path, 'w', encoding='utf-8') as f:
f.write(encrypted_content)
# Delete original unencrypted file
os.remove(file_path)
logger.info(f"Encrypted file: {filepath}")
except Exception as e:
logger.error(f"Error encrypting file {filepath}: {e}")
return jsonify({'success': False, 'error': f"Failed to encrypt file: {e}"}), 500
logger.info(f"Protected file: {file_key}")
elif action == 'unprotect':
if file_key in passwords:
del passwords[file_key]
logger.info(f"Unprotected file: {file_key}")
else:
return jsonify({'success': False, 'error': 'Invalid action'}), 400
if save_passwords(passwords):
return jsonify({'success': True, 'message': f"File {action}ed successfully"})
else:
return jsonify({'success': False, 'error': 'Failed to save password'}), 500
@app.route('/api/status')
def status():
"""Return application status."""
repos = load_repositories()
repo_status = {}
for repo_name in repos.keys():
repo_path = os.path.join(REPOS_DIR, repo_name)
repo_status[repo_name] = {
'cloned': os.path.exists(repo_path),
'last_modified': datetime.fromtimestamp(
os.path.getmtime(repo_path)
).isoformat() if os.path.exists(repo_path) else None
}
content_status = {
'exists': os.path.exists(CONTENT_DIR),
'last_modified': datetime.fromtimestamp(
os.path.getmtime(CONTENT_DIR)
).isoformat() if os.path.exists(CONTENT_DIR) else None
}
return jsonify({
'status': 'running',
'repositories': repo_status,
'content': content_status,
'encryption_enabled': ENCRYPTION_ENABLED,
'timestamp': datetime.now().isoformat()
})
if __name__ == '__main__':
# Perform initial pull and index
# Perform initial index build
logger.info("Initializing markmywords")
scheduled_pull()
# Setup scheduler
scheduler = setup_scheduler()
update_search_index()
# Start Flask app
app.run(host='0.0.0.0', port=5000, debug=False)