#!/usr/bin/env python3 """Local authoring UI for the Org published website.""" from __future__ import annotations import json import os import posixpath import re import struct import subprocess import sys import threading import time import cgi from dataclasses import dataclass, field from datetime import datetime from email.utils import formatdate from http import HTTPStatus from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path from typing import Any from urllib.parse import parse_qs, urlparse def looks_like_content_root(path: Path) -> bool: return (path / "authoring_server.py").exists() and ( (path / "blogs").exists() or (path / "posts").exists() or (path / "lima").exists() ) def resolve_root() -> Path: env_root = os.environ.get("AUTHOR_ROOT") if env_root: return Path(env_root).expanduser().resolve() candidates = [ Path.cwd(), Path(__file__).resolve().parent, ] workspace = os.environ.get("GITHUB_WORKSPACE") if workspace: candidates.insert(0, Path(workspace)) for base in list(candidates): candidates.extend(base.parents) seen = set() for candidate in candidates: resolved = candidate.expanduser().resolve() if resolved in seen: continue seen.add(resolved) if looks_like_content_root(resolved): return resolved return Path(__file__).resolve().parent ROOT = resolve_root() BLOGS_DIR = ROOT / "blogs" POSTS_DIR = ROOT / "posts" LIMA_DIR = ROOT / "lima" IMAGE_ASSETS_DIR = ROOT / "assets" / "images" HZONE_ASSETS_DIR = IMAGE_ASSETS_DIR / "hzone" EXCLUDED_CONTENT_DIR_NAMES = { ".agents", ".codex", ".git", ".packages", ".venv", "__pycache__", "assets", "backups", "output", "tags", } GENERATED_ORG_NAMES = { "blogs-list.org", "books-list.org", "posts-list.org", "career-list.org", "sitemap.org", "recently-updated.org", "wip.org", } GENERATED_CONTENT_NAMES = GENERATED_ORG_NAMES | {"lima-list.org"} ALLOWED_UPLOAD_EXTENSIONS = { ".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg", } MONTH_NAMES = [ "january", "february", "march", "april", "may", "june", "july", "august", "september", "october", "november", "december", ] def slugify(value: str) -> str: slug = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-") return slug or "untitled" def normalise_tags(value: Any) -> list[str]: if isinstance(value, str): raw = re.split(r"[,:\s]+", value) elif isinstance(value, list): raw = [str(item) for item in value] else: raw = [] tags = [] for tag in raw: if not tag.strip(): continue clean = slugify(tag) if clean and clean not in tags: tags.append(clean) return tags def org_date(dt: datetime) -> str: return dt.strftime("<%Y-%m-%d %a %H:%M>") def parse_org_datetime(value: str | None) -> datetime | None: if not value: return None match = re.search(r"<(\d{4})-(\d{2})-(\d{2})(?:\s+\w+)?(?:\s+(\d{2}):(\d{2}))?>", value) if not match: return None year, month, day, hour, minute = match.groups() return datetime( int(year), int(month), int(day), int(hour or 12), int(minute or 0), ) def html_escape(value: str) -> str: return ( value.replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """) ) @dataclass class OrgPage: path: str page_type: str title: str slug: str tags: list[str] content: str date: str comments: bool options: str wip: str | None = None @dataclass class ContentPage: path: str page_type: str title: str slug: str tags: list[str] content: str date: str comments: bool options: str format: str wip: str | None = None @dataclass class BuildJob: id: int path: str title: str queued_at: float = field(default_factory=time.time) started_at: float | None = None finished_at: float | None = None ok: bool | None = None message: str = "Queued" log: str = "" @property def status(self) -> str: if self.finished_at is not None: return "done" if self.ok else "failed" if self.started_at is not None: return "running" return "queued" def to_dict(self, include_log: bool = False) -> dict[str, Any]: data = { "id": self.id, "path": self.path, "title": self.title, "queuedAt": self.queued_at, "startedAt": self.started_at, "finishedAt": self.finished_at, "ok": self.ok, "status": self.status, "message": self.message, } if include_log: data["log"] = self.log[-12000:] return data class BuildQueue: def __init__(self) -> None: self._lock = threading.Lock() self._next_id = 1 self._pending: list[BuildJob] = [] self._current: BuildJob | None = None self._recent: list[BuildJob] = [] self._worker: threading.Thread | None = None def snapshot(self) -> dict[str, Any]: with self._lock: current = self._current.to_dict(include_log=True) if self._current else None recent = [job.to_dict(include_log=True) for job in self._recent[-10:]] pending = [job.to_dict() for job in self._pending] latest = current or (recent[-1] if recent else None) message = latest["message"] if latest else "No builds have run yet." return { "running": current is not None, "queued": len(pending), "message": message, "current": current, "pending": pending, "recent": recent, "log": latest.get("log", "") if latest else "", } def enqueue(self, path: str, title: str) -> BuildJob: with self._lock: job = BuildJob(self._next_id, path, title) self._next_id += 1 self._pending.append(job) if self._worker is None or not self._worker.is_alive(): self._worker = threading.Thread(target=self._run_worker, daemon=True) self._worker.start() return job def _run_worker(self) -> None: while True: with self._lock: if not self._pending: self._current = None return job = self._pending.pop(0) job.started_at = time.time() job.message = "Publishing site and search index." self._current = job ok, message, log = run_build_commands() with self._lock: job.finished_at = time.time() job.ok = ok job.message = message job.log = log self._recent.append(job) self._recent = self._recent[-20:] self._current = None BUILD_QUEUE = BuildQueue() def safe_relative_path(path: str) -> Path: rel = Path(path) if rel.is_absolute() or ".." in rel.parts: raise ValueError("Path must stay inside this repository.") full = (ROOT / rel).resolve() if not full.is_relative_to(ROOT): raise ValueError("Path must stay inside this repository.") if full.name in GENERATED_CONTENT_NAMES or "sync-conflict" in full.name: raise ValueError("Generated and sync-conflict files are not editable here.") rel_parts = full.relative_to(ROOT).parts if any(part in EXCLUDED_CONTENT_DIR_NAMES for part in rel_parts): raise ValueError("This path is outside the editable content folders.") if full.suffix == ".org": return full if full.suffix == ".md" and full.is_relative_to(LIMA_DIR): return full raise ValueError("Only .org content files and .md files under lima can be edited here.") def safe_target_path(path: str, slug: str, page_type: str) -> Path: candidate = path.strip() if not candidate: raise ValueError("Path is required.") default_ext = ".md" if page_type == "lima" else ".org" if candidate.endswith("/"): candidate = f"{candidate}{slug}{default_ext}" elif not Path(candidate).suffix: candidate = f"{candidate}{default_ext}" return safe_relative_path(candidate) def markdown_title(content: str, fallback: str) -> str: for line in content.splitlines(): match = re.match(r"^#{1,6}\s+(.+?)\s*$", line) if match: return match.group(1).strip() return fallback.replace("-", " ").replace("_", " ").title() def read_markdown_page(path: Path) -> ContentPage: content = path.read_text(encoding="utf-8") rel = path.relative_to(ROOT).as_posix() title = markdown_title(content, path.stem) return ContentPage( path=rel, page_type="lima", title=title, slug=path.stem, tags=[], content=content, date="", comments=True, options="", format="markdown", ) def read_page(path: Path) -> ContentPage: if path.suffix == ".md" and path.is_relative_to(LIMA_DIR): return read_markdown_page(path) text = path.read_text(encoding="utf-8") meta: dict[str, str] = {} body_lines: list[str] = [] in_header = True for line in text.splitlines(): if in_header and line.startswith("#+"): key, _, value = line[2:].partition(":") meta[key.strip().upper()] = value.strip() else: in_header = False body_lines.append(line) rel = path.relative_to(ROOT).as_posix() if path.is_relative_to(BLOGS_DIR): page_type = "blog" elif path.is_relative_to(POSTS_DIR): page_type = "post" else: page_type = "page" slug = meta.get("SLUG") or path.stem tags = normalise_tags(meta.get("FILETAGS", "")) return ContentPage( path=rel, page_type=page_type, title=meta.get("TITLE", path.stem), slug=slug, tags=tags, content="\n".join(body_lines).lstrip("\n"), date=meta.get("DATE", org_date(datetime.fromtimestamp(path.stat().st_mtime))), comments=meta.get("COMMENTS", "t").lower() == "t", options=meta.get("OPTIONS", "num:nil"), format="org", wip=meta.get("WIP"), ) def page_to_dict(page: ContentPage) -> dict[str, Any]: return { "path": page.path, "pageType": page.page_type, "title": page.title, "slug": page.slug, "tags": page.tags, "content": page.content, "date": page.date, "comments": page.comments, "options": page.options, "format": page.format, "wip": page.wip or "", } def list_pages() -> list[dict[str, Any]]: pages = [] org_paths = [] if ROOT.exists(): try: for path in ROOT.rglob("*.org"): try: rel_parts = path.relative_to(ROOT).parts except ValueError: continue if any(part in EXCLUDED_CONTENT_DIR_NAMES for part in rel_parts): continue org_paths.append(path) except OSError: org_paths = [] try: md_paths = list(LIMA_DIR.rglob("*.md")) if LIMA_DIR.exists() else [] except OSError: md_paths = [] for path in sorted(org_paths + md_paths): if path.name in GENERATED_CONTENT_NAMES or "sync-conflict" in path.name: continue try: page = read_page(path) parsed = parse_org_datetime(page.date) timestamp = parsed.timestamp() if parsed else path.stat().st_mtime except (OSError, UnicodeDecodeError, ValueError): continue pages.append( { "path": page.path, "pageType": page.page_type, "title": page.title, "slug": page.slug, "tags": page.tags, "date": page.date, "format": page.format, "timestamp": timestamp, } ) return sorted(pages, key=lambda item: item["timestamp"], reverse=True) def target_path(data: dict[str, Any], existing_path: str | None) -> Path: if existing_path: return safe_relative_path(existing_path) title = str(data.get("title") or "").strip() slug = slugify(str(data.get("slug") or title)) page_type = str(data.get("pageType") or "blog") explicit_path = str(data.get("targetPath") or "").strip() if explicit_path: return safe_target_path(explicit_path, slug, page_type) if page_type == "blog": dt = parse_org_datetime(str(data.get("date") or "")) or datetime.now() folder = BLOGS_DIR / str(dt.year) / f"{dt.month:02d}-{MONTH_NAMES[dt.month - 1]}" return folder / f"{slug}.org" if page_type == "post": raw_section = str(data.get("section") or "").strip() section = slugify(raw_section) if raw_section else "" folder = POSTS_DIR / section if section else POSTS_DIR return folder / f"{slug}.org" if page_type == "lima": return LIMA_DIR / f"{slug}.md" if page_type == "page": return ROOT / f"{slug}.org" raise ValueError("pageType must be blog, post, page, or lima.") def render_markdown(data: dict[str, Any]) -> str: content = str(data.get("content") or "").replace("\r\n", "\n").strip() title = str(data.get("title") or "").strip() if not title: raise ValueError("Title is required.") content = re.sub( r']*>\s*]*\bsrc="([^"]+)"[^>]*\balt="([^"]*)"[^>]*>\s*', lambda match: f"![{match.group(2)}]({match.group(1)})", content, flags=re.IGNORECASE, ) if content: if re.search(r"^#{1,6}[^\S\r\n]+.+?[^\S\r\n]*$", content, flags=re.MULTILINE): content = re.sub( r"^#{1,6}[^\S\r\n]+.+?[^\S\r\n]*$", f"# {title}", content, count=1, flags=re.MULTILINE, ) else: content = f"# {title}\n\n{content}" return content + "\n" return f"# {title}\n" def render_org(data: dict[str, Any], previous: ContentPage | None) -> str: title = str(data.get("title") or "").strip() if not title: raise ValueError("Title is required.") slug = slugify(str(data.get("slug") or title)) tags = normalise_tags(data.get("tags", [])) content = str(data.get("content") or "").replace("\r\n", "\n").strip() date = str(data.get("date") or "").strip() if not parse_org_datetime(date): date = previous.date if previous else org_date(datetime.now()) options = str(data.get("options") or (previous.options if previous else "num:nil")).strip() comments = bool(data.get("comments", True)) lines = [ f"#+TITLE: {title}", f"#+OPTIONS: {options}", f"#+DATE: {date}", f"#+filetags: {''.join(f':{tag}' for tag in tags)}:", ] wip = str(data.get("wip") or (previous.wip if previous else "") or "").strip() if wip: lines.append(f"#+WIP: {wip}") lines.extend( [ f"#+COMMENTS: {'t' if comments else ''}", f"#+SLUG: {slug}", "", content, "", ] ) return "\n".join(lines) def save_page(data: dict[str, Any]) -> dict[str, Any]: existing_path = data.get("path") or None target = target_path(data, str(existing_path) if existing_path else None) previous = read_page(target) if target.exists() else None if not target.parent.exists(): target.parent.mkdir(parents=True) if target.exists() and not existing_path: raise ValueError(f"{target.relative_to(ROOT)} already exists.") if target.suffix == ".md": target.write_text(render_markdown(data), encoding="utf-8") else: target.write_text(render_org(data, previous), encoding="utf-8") return page_to_dict(read_page(target)) def image_dimensions(payload: bytes, ext: str) -> tuple[int, int] | None: if ext == ".png" and payload.startswith(b"\x89PNG\r\n\x1a\n") and len(payload) >= 24: width, height = struct.unpack(">II", payload[16:24]) return width, height if ext == ".gif" and payload[:6] in {b"GIF87a", b"GIF89a"} and len(payload) >= 10: width, height = struct.unpack(" len(payload): break size = int.from_bytes(payload[i:i + 2], "big") if size < 2: break if marker in {0xC0, 0xC1, 0xC2, 0xC3, 0xC5, 0xC6, 0xC7, 0xC9, 0xCA, 0xCB, 0xCD, 0xCE, 0xCF}: if i + 7 <= len(payload): height = int.from_bytes(payload[i + 3:i + 5], "big") width = int.from_bytes(payload[i + 5:i + 7], "big") return width, height break i += size return None def relative_asset_path(page_path: str, asset_path: str) -> str: page = safe_relative_path(page_path) if page_path else ROOT / "index.org" page_rel = page.relative_to(ROOT).as_posix() page_output_dir = posixpath.dirname(page_rel) return posixpath.relpath(asset_path, page_output_dir or ".") def gallery_image_html(filename: str, asset_path: str, page_path: str, payload: bytes, ext: str) -> str: absolute_url = f"https://zainezq.com/{asset_path}" relative_url = relative_asset_path(page_path, asset_path) dims = image_dimensions(payload, ext) width, height = dims if dims else (1920, 1080) alt = html_escape(filename) return ( f'' f'{alt}' ) def attachment_image_dir(page_path: str) -> Path: page = safe_relative_path(page_path) if page_path else ROOT / "index.org" if page.suffix == ".md" and page.is_relative_to(LIMA_DIR): return HZONE_ASSETS_DIR if page.is_relative_to(POSTS_DIR): rel = page.relative_to(POSTS_DIR) section = rel.parts[0] if len(rel.parts) > 1 else "posts" return IMAGE_ASSETS_DIR / slugify(section) if page.is_relative_to(BLOGS_DIR): return IMAGE_ASSETS_DIR / "blogs" rel = page.relative_to(ROOT) if len(rel.parts) > 1: return IMAGE_ASSETS_DIR / slugify(rel.parts[0]) return IMAGE_ASSETS_DIR / "pages" def save_upload(filename: str, payload: bytes, page_path: str = "") -> dict[str, str]: original = Path(filename or "attachment").name ext = Path(original).suffix.lower() if ext not in ALLOWED_UPLOAD_EXTENSIONS: raise ValueError("Only common image files can be uploaded.") now = datetime.now() target_dir = attachment_image_dir(page_path) target_dir.mkdir(parents=True, exist_ok=True) stem = slugify(Path(original).stem) prefix = "" if re.match(r"^\d{4}-\d{2}-\d{2}-", stem) else f"{now.strftime('%Y-%m-%d')}-" target = target_dir / f"{prefix}{stem}{ext}" counter = 2 while target.exists(): target = target_dir / f"{prefix}{stem}-{counter}{ext}" counter += 1 target.write_bytes(payload) rel = target.relative_to(ROOT).as_posix() absolute_url = f"https://zainezq.com/{rel}" relative_url = relative_asset_path(page_path, rel) is_markdown = page_path.endswith(".md") insert_text = f"![{target.name}]({relative_url})" if is_markdown else f"[[{relative_url}]]" return { "url": absolute_url, "relativeUrl": relative_url, "path": rel, "markdown": insert_text, "insertText": insert_text, "filename": target.name, } def run_build_commands() -> tuple[bool, str, str]: venv_python = ROOT / ".venv" / "bin" / "python" venv_pip = ROOT / ".venv" / "bin" / "pip" commands = [["emacs", "-Q", "--script", "build-site.el"]] if not venv_python.exists(): commands.extend( [ [sys.executable, "-m", "venv", ".venv"], [str(venv_pip), "install", "-r", "requirements.txt"], ] ) commands.append([str(venv_python), "search-index-json.py"]) combined = [] ok = True for command in commands: combined.append(f"$ {' '.join(command)}\n") proc = subprocess.run( command, cwd=ROOT, text=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, ) combined.append(proc.stdout) if proc.returncode != 0: ok = False combined.append(f"\nCommand exited with {proc.returncode}.\n") break message = "Build complete. The site output and search index were regenerated." if ok else "Build failed. Check the log below." return ok, message, "".join(combined) def queue_build(page: dict[str, Any]) -> dict[str, Any]: return BUILD_QUEUE.enqueue(page["path"], page["title"]).to_dict() APP_HTML = r""" Org Site Authoring

New page

Build queue

Latest build log


    
""" class Handler(BaseHTTPRequestHandler): server_version = "OrgAuthoring/1.0" def log_message(self, fmt: str, *args: Any) -> None: sys.stderr.write("%s - %s\n" % (formatdate(time.time()), fmt % args)) def send_json(self, data: Any, status: HTTPStatus = HTTPStatus.OK) -> None: body = json.dumps(data).encode("utf-8") self.send_response(status) self.send_header("Content-Type", "application/json; charset=utf-8") self.send_header("Content-Length", str(len(body))) self.send_header("Cache-Control", "no-store") self.end_headers() self.wfile.write(body) def do_GET(self) -> None: parsed = urlparse(self.path) if parsed.path == "/": body = APP_HTML.encode("utf-8") self.send_response(HTTPStatus.OK) self.send_header("Content-Type", "text/html; charset=utf-8") self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) return if parsed.path == "/api/pages": try: self.send_json(list_pages()) except Exception as exc: self.send_json({"error": str(exc)}, HTTPStatus.INTERNAL_SERVER_ERROR) return if parsed.path == "/api/page": query = parse_qs(parsed.query) try: path = safe_relative_path(query.get("path", [""])[0]) self.send_json(page_to_dict(read_page(path))) except Exception as exc: self.send_json({"error": str(exc)}, HTTPStatus.BAD_REQUEST) return if parsed.path == "/api/build": self.send_json(BUILD_QUEUE.snapshot()) return self.send_error(HTTPStatus.NOT_FOUND) def do_POST(self) -> None: if self.path == "/api/upload": try: form = cgi.FieldStorage( fp=self.rfile, headers=self.headers, environ={ "REQUEST_METHOD": "POST", "CONTENT_TYPE": self.headers.get("Content-Type", ""), }, ) field = form["attachment"] if "attachment" in form else None if field is None or not getattr(field, "filename", ""): raise ValueError("No attachment was uploaded.") payload = field.file.read() if not payload: raise ValueError("Attachment is empty.") page_path = "" if "pagePath" in form: page_path = str(form["pagePath"].value or "") self.send_json(save_upload(field.filename, payload, page_path)) except Exception as exc: self.send_json({"error": str(exc)}, HTTPStatus.BAD_REQUEST) return if self.path != "/api/page": self.send_error(HTTPStatus.NOT_FOUND) return try: length = int(self.headers.get("Content-Length", "0")) data = json.loads(self.rfile.read(length).decode("utf-8")) saved = save_page(data) saved["queuedBuild"] = queue_build(saved) self.send_json(saved) except Exception as exc: self.send_json({"error": str(exc)}, HTTPStatus.BAD_REQUEST) def main() -> None: port = int(os.environ.get("AUTHOR_PORT", "8765")) server = ThreadingHTTPServer(("127.0.0.1", port), Handler) print(f"Authoring UI running at http://127.0.0.1:{port}") print(f"Content root: {ROOT}") print("Press Ctrl-C to stop.") try: server.serve_forever() except KeyboardInterrupt: pass if __name__ == "__main__": main()