Updated to work a lot better.

This commit is contained in:
Discsearcher
2026-08-16 15:00:44 -04:00
parent 85e6af0b7d
commit d035053203
6 changed files with 185 additions and 48 deletions
+118 -13
View File
@@ -8,11 +8,12 @@ import subprocess
import json
from pathlib import Path
from datetime import datetime
from flask import Flask, render_template, request, jsonify
from flask import Flask, render_template, request, jsonify, send_file
from apscheduler.schedulers.background import BackgroundScheduler
from apscheduler.triggers.cron import CronTrigger
import markdown
import logging
import re
from whoosh.index import create_in, open_dir
from whoosh.fields import Schema, TEXT, ID
from whoosh.qparser import QueryParser
@@ -48,6 +49,88 @@ def create_search_index():
ix = create_search_index()
def is_hidden_path(file_path):
"""Check if a file path contains any hidden directories or files (starting with dot)."""
path_obj = Path(file_path) if isinstance(file_path, str) else file_path
# Check all parts of the path including the filename
return any(part.startswith('.') for part in path_obj.parts)
def rewrite_image_paths(md_content, repo, filepath):
"""Rewrite relative image paths to be served by Flask."""
# Get the directory of the current file (relative to repo root)
file_dir = os.path.dirname(filepath)
# Pattern to match markdown image syntax: ![alt](path)
def replace_image_path(match):
alt_text = match.group(1)
img_path = match.group(2)
# Skip absolute URLs and data URIs
if img_path.startswith(('http://', 'https://', 'data:')):
return match.group(0)
# Resolve relative paths
if img_path.startswith('/'):
# Absolute path from repo root
resolved_path = img_path.lstrip('/')
else:
# Relative path - resolve it relative to the file's directory
if file_dir:
resolved_path = os.path.normpath(os.path.join(file_dir, img_path))
else:
resolved_path = img_path
# Ensure forward slashes for URL
resolved_path = resolved_path.replace(os.sep, '/')
image_url = f"/image/{repo}/{resolved_path}"
return f"![{alt_text}]({image_url})"
# Replace all markdown image references
md_content = re.sub(r'!\[([^\]]*)\]\(([^\)]+)\)', replace_image_path, md_content)
return md_content
def build_directory_tree(repo_path):
"""Build a hierarchical tree of markdown files organized by directory."""
tree = {}
if not os.path.exists(repo_path):
logger.warning(f"Repo path does not exist: {repo_path}")
return tree
file_count = 0
for md_file in Path(repo_path).rglob('*'):
if md_file.suffix.lower() not in MARKDOWN_EXTENSIONS:
continue
if is_hidden_path(md_file):
logger.debug(f"Skipping hidden path: {md_file}")
continue
file_count += 1
rel_path = md_file.relative_to(repo_path)
parts = rel_path.parts[:-1] # All parts except filename
filename = md_file.stem # Name without extension
logger.debug(f"Adding to tree: {rel_path} (name: {filename})")
# Navigate/create nested dict structure
current = tree
for part in parts:
if part not in current:
current[part] = {}
current = current[part]
# Add file to current level
if '_files' not in current:
current['_files'] = []
current['_files'].append({
'name': filename,
'path': str(rel_path)
})
logger.info(f"Built tree for {repo_path}: found {file_count} markdown files")
return tree
def load_repositories():
"""Load repository configuration from file."""
if not os.path.exists(CONFIG_FILE):
@@ -115,7 +198,7 @@ def update_search_index():
continue
for md_file in repo_dir.rglob('*'):
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS:
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS and not is_hidden_path(md_file):
try:
content = md_file.read_text(encoding='utf-8', errors='ignore')
# Extract title from filename or first heading
@@ -181,24 +264,43 @@ def index():
for repo_name in repos.keys():
repo_path = os.path.join(REPOS_DIR, repo_name)
markdown_files = []
if os.path.exists(repo_path):
for md_file in Path(repo_path).rglob('*'):
if md_file.suffix.lower() in MARKDOWN_EXTENSIONS:
rel_path = str(md_file.relative_to(repo_path))
markdown_files.append({
'name': md_file.stem,
'path': rel_path
})
file_tree = build_directory_tree(repo_path)
repo_list.append({
'name': repo_name,
'files': sorted(markdown_files, key=lambda x: x['path'])
'tree': file_tree
})
return render_template('index.html', repositories=repo_list)
@app.route('/image/<repo>/<path:filepath>')
def serve_image(repo, filepath):
"""Serve images from repository directories."""
# Security: prevent directory traversal
if '..' in filepath or filepath.startswith('/'):
return "Invalid path", 400
file_path = os.path.join(REPOS_DIR, repo, filepath)
# Ensure the file is within the repo directory
try:
file_path = os.path.realpath(file_path)
repo_path = os.path.realpath(os.path.join(REPOS_DIR, repo))
if not file_path.startswith(repo_path):
return "Access denied", 403
except Exception:
return "Invalid path", 400
if not os.path.exists(file_path):
logger.warning(f"Image not found: {file_path}")
return "Image not found", 404
try:
return send_file(file_path)
except Exception as e:
logger.error(f"Error serving image {file_path}: {e}")
return f"Error serving image: {e}", 500
@app.route('/api/search', methods=['GET'])
def search():
"""Search markdown files."""
@@ -249,6 +351,9 @@ def view_file(repo, filepath):
with open(file_path, 'r', encoding='utf-8') as f:
md_content = f.read()
# Rewrite image paths to be served by Flask
md_content = rewrite_image_paths(md_content, repo, filepath)
# Convert markdown to HTML
html_content = markdown.markdown(
md_content,