#!/usr/bin/env python3 """Build bounded JSON cache chunks for a Markdown vault.""" from __future__ import annotations import json import os import re import sys import tempfile import unicodedata from pathlib import Path from typing import Any SKIP_DIRS = { ".git", ".obsidian", ".trash", ".stfolder", "node_modules", "dist", "build", } HEADING = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE) WORDS = re.compile(r"[^\W_]+", re.UNICODE) CHUNK_BYTES = 96 * 1024 RECORD_BYTES = 48 * 1024 def visible_directory(name: str) -> bool: return name not in SKIP_DIRS and not name.startswith(".") def note_title(path: Path, content: str) -> str: match = HEADING.search(content) return match.group(1).strip() if match else path.stem def folded(value: str) -> str: """Case-fold and strip accents so ASCII queries match Unicode notes.""" normalized = unicodedata.normalize("NFKD", value.casefold()) return "".join( character for character in normalized if not unicodedata.combining(character) ) def token_buckets(content: str) -> dict[str, dict[str, list[str]]]: buckets: dict[str, dict[str, list[str]]] = {} for token in sorted(set(WORDS.findall(content))): if not token: continue by_length = buckets.setdefault(token[0], {}) by_length.setdefault(str(len(token)), []).append(token) return buckets def encoded(record: dict[str, Any]) -> str: return json.dumps(record, ensure_ascii=False, separators=(",", ":")) def fold_records(characters: set[str]) -> list[dict[str, Any]]: mapping: dict[str, str] = {} for character in characters: variants = {character, character.lower(), character.upper(), character.title()} for variant in variants: for variant_character in variant: mapping[variant_character] = folded(variant_character) records: list[dict[str, Any]] = [] current: dict[str, str] = {} for source, target in sorted(mapping.items()): candidate = {**current, source: target} record = {"kind": "fold", "map": candidate} if current and len(encoded(record).encode("utf-8")) > RECORD_BYTES: records.append({"kind": "fold", "map": current}) current = {source: target} else: current = candidate if current: records.append({"kind": "fold", "map": current}) return records def directory_record(relative: str) -> dict[str, Any]: record = {"kind": "directory", "relative": relative} if len(encoded(record).encode("utf-8")) > RECORD_BYTES: raise ValueError(f"directory metadata exceeded bounded size: {relative}") return record def note_record(title: str, relative: str) -> dict[str, Any]: display_title = title[:256] + ("…" if len(title) > 256 else "") record = { "kind": "note", "id": relative, "title": display_title, "relative": relative, "titleRaw": display_title.casefold(), "pathRaw": relative.casefold(), "titleSearch": folded(display_title), "pathSearch": folded(relative), } if len(encoded(record).encode("utf-8")) > RECORD_BYTES: raise ValueError(f"note metadata exceeded bounded size: {relative}") return record def content_record(relative: str, fragment: str) -> dict[str, Any]: content_search = folded(fragment) return { "kind": "content", "id": relative, "fragment": fragment, "lines": fragment.split("\n"), "contentRaw": fragment.casefold(), "contentSearch": content_search, "contentTokens": token_buckets(content_search), } def content_records(relative: str, content: str) -> list[dict[str, Any]]: """Split one note into bounded records without changing its original text.""" pending = [content] records: list[dict[str, Any]] = [] while pending: fragment = pending.pop() record = content_record(relative, fragment) if len(encoded(record).encode("utf-8")) <= RECORD_BYTES: records.append(record) continue midpoint = max(len(fragment) // 2, 1) if midpoint >= len(fragment): raise ValueError(f"cannot split oversized content in {relative}") pending.extend((fragment[midpoint:], fragment[:midpoint])) return records def index_records(root: Path) -> list[dict[str, Any]]: notes: list[tuple[str, str, str]] = [] indexed_directories: set[str] = set() characters: set[str] = set() for current, directories, filenames in os.walk(root): directories[:] = sorted( name for name in directories if visible_directory(name) and not Path(current, name).is_symlink() ) for directory in directories: relative_directory = Path(current, directory).relative_to(root).as_posix() indexed_directories.add(relative_directory) characters.update(relative_directory) for filename in sorted(filenames): if not filename.lower().endswith(".md") or filename.startswith("."): continue path = Path(current, filename) if path.is_symlink(): continue try: content = path.read_text(encoding="utf-8", errors="replace") except OSError: continue relative = path.relative_to(root).as_posix() title = note_title(path, content) characters.update(title) characters.update(relative) characters.update(content) notes.append((title, relative, content)) notes.sort(key=lambda note: (note[0].casefold(), note[1].casefold())) records = fold_records(characters) for relative in sorted(indexed_directories, key=str.casefold): records.append(directory_record(relative)) for title, relative, content in notes: records.append(note_record(title, relative)) records.extend(content_records(relative, content)) return records def write_chunks(cache: Path, records: list[dict[str, Any]]) -> list[str]: encoded_records = [encoded(record) for record in records] chunks: list[list[str]] = [] current: list[str] = [] current_bytes = 2 for record in encoded_records: record_bytes = len(record.encode("utf-8")) if record_bytes > RECORD_BYTES: raise ValueError("index record exceeded its bounded size") separator_bytes = 1 if current else 0 if current and current_bytes + separator_bytes + record_bytes > CHUNK_BYTES: chunks.append(current) current = [] current_bytes = 2 separator_bytes = 0 current.append(record) current_bytes += separator_bytes + record_bytes if current or not chunks: chunks.append(current) names: list[str] = [] for index, chunk in enumerate(chunks): chunk_name = f"index-cache-{index:03d}.json" path = (cache.parent / chunk_name).resolve() if path.parent != cache.parent: raise ValueError("cache chunk escaped the plugin directory") with tempfile.NamedTemporaryFile( mode="w", encoding="utf-8", dir=cache.parent, delete=False ) as handle: handle.write("[" + ",".join(chunk) + "]") temporary = Path(handle.name) os.replace(temporary, path) names.append(chunk_name) active = set(names) for stale in cache.parent.glob(f"{cache.stem}-*.json"): if stale.name not in active: stale.unlink(missing_ok=True) cache.unlink(missing_ok=True) return names def main() -> int: if len(sys.argv) != 3: print("usage: index.py VAULT CACHE", file=sys.stderr) return 2 root = Path(sys.argv[1]).expanduser().resolve() requested_cache = Path(sys.argv[2]).expanduser().resolve() plugin_dir = Path(__file__).resolve().parent if ( requested_cache.parent != plugin_dir or requested_cache.name != "index-cache.json" ): print("cache must be index-cache.json in the plugin directory", file=sys.stderr) return 2 if not root.is_dir(): print(f"vault does not exist: {root}", file=sys.stderr) return 1 records = index_records(root) chunk_names = write_chunks(plugin_dir / "index-cache.json", records) note_count = sum(record["kind"] == "note" for record in records) json.dump( {"chunks": chunk_names, "notes": note_count}, sys.stdout, separators=(",", ":"), ) return 0 if __name__ == "__main__": raise SystemExit(main())