Files
alex ed063ab1fe feat(desktop): consolidate declarative workstation profile
Capture desktop-specific system, session, Noctalia, and Pi integration alongside shared laptop support. Retire obsolete X11 configuration and make package ownership explicit for repeatable Decman deployments.
2026-08-24 20:20:58 +02:00

251 lines
8.5 KiB
Python

#!/usr/bin/env python3
"""Build bounded JSON cache chunks for a Markdown vault."""
from __future__ import annotations
import json
import os
import re
import sys
import tempfile
import unicodedata
from pathlib import Path
from typing import Any
SKIP_DIRS = {
".git",
".obsidian",
".trash",
".stfolder",
"node_modules",
"dist",
"build",
}
HEADING = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE)
WORDS = re.compile(r"[^\W_]+", re.UNICODE)
CHUNK_BYTES = 96 * 1024
RECORD_BYTES = 48 * 1024
def visible_directory(name: str) -> bool:
return name not in SKIP_DIRS and not name.startswith(".")
def note_title(path: Path, content: str) -> str:
match = HEADING.search(content)
return match.group(1).strip() if match else path.stem
def folded(value: str) -> str:
"""Case-fold and strip accents so ASCII queries match Unicode notes."""
normalized = unicodedata.normalize("NFKD", value.casefold())
return "".join(
character for character in normalized if not unicodedata.combining(character)
)
def token_buckets(content: str) -> dict[str, dict[str, list[str]]]:
buckets: dict[str, dict[str, list[str]]] = {}
for token in sorted(set(WORDS.findall(content))):
if not token:
continue
by_length = buckets.setdefault(token[0], {})
by_length.setdefault(str(len(token)), []).append(token)
return buckets
def encoded(record: dict[str, Any]) -> str:
return json.dumps(record, ensure_ascii=False, separators=(",", ":"))
def fold_records(characters: set[str]) -> list[dict[str, Any]]:
mapping: dict[str, str] = {}
for character in characters:
variants = {character, character.lower(), character.upper(), character.title()}
for variant in variants:
for variant_character in variant:
mapping[variant_character] = folded(variant_character)
records: list[dict[str, Any]] = []
current: dict[str, str] = {}
for source, target in sorted(mapping.items()):
candidate = {**current, source: target}
record = {"kind": "fold", "map": candidate}
if current and len(encoded(record).encode("utf-8")) > RECORD_BYTES:
records.append({"kind": "fold", "map": current})
current = {source: target}
else:
current = candidate
if current:
records.append({"kind": "fold", "map": current})
return records
def directory_record(relative: str) -> dict[str, Any]:
record = {"kind": "directory", "relative": relative}
if len(encoded(record).encode("utf-8")) > RECORD_BYTES:
raise ValueError(f"directory metadata exceeded bounded size: {relative}")
return record
def note_record(title: str, relative: str) -> dict[str, Any]:
display_title = title[:256] + ("…" if len(title) > 256 else "")
record = {
"kind": "note",
"id": relative,
"title": display_title,
"relative": relative,
"titleRaw": display_title.casefold(),
"pathRaw": relative.casefold(),
"titleSearch": folded(display_title),
"pathSearch": folded(relative),
}
if len(encoded(record).encode("utf-8")) > RECORD_BYTES:
raise ValueError(f"note metadata exceeded bounded size: {relative}")
return record
def content_record(relative: str, fragment: str) -> dict[str, Any]:
content_search = folded(fragment)
return {
"kind": "content",
"id": relative,
"fragment": fragment,
"lines": fragment.split("\n"),
"contentRaw": fragment.casefold(),
"contentSearch": content_search,
"contentTokens": token_buckets(content_search),
}
def content_records(relative: str, content: str) -> list[dict[str, Any]]:
"""Split one note into bounded records without changing its original text."""
pending = [content]
records: list[dict[str, Any]] = []
while pending:
fragment = pending.pop()
record = content_record(relative, fragment)
if len(encoded(record).encode("utf-8")) <= RECORD_BYTES:
records.append(record)
continue
midpoint = max(len(fragment) // 2, 1)
if midpoint >= len(fragment):
raise ValueError(f"cannot split oversized content in {relative}")
pending.extend((fragment[midpoint:], fragment[:midpoint]))
return records
def index_records(root: Path) -> list[dict[str, Any]]:
notes: list[tuple[str, str, str]] = []
indexed_directories: set[str] = set()
characters: set[str] = set()
for current, directories, filenames in os.walk(root):
directories[:] = sorted(
name
for name in directories
if visible_directory(name) and not Path(current, name).is_symlink()
)
for directory in directories:
relative_directory = Path(current, directory).relative_to(root).as_posix()
indexed_directories.add(relative_directory)
characters.update(relative_directory)
for filename in sorted(filenames):
if not filename.lower().endswith(".md") or filename.startswith("."):
continue
path = Path(current, filename)
if path.is_symlink():
continue
try:
content = path.read_text(encoding="utf-8", errors="replace")
except OSError:
continue
relative = path.relative_to(root).as_posix()
title = note_title(path, content)
characters.update(title)
characters.update(relative)
characters.update(content)
notes.append((title, relative, content))
notes.sort(key=lambda note: (note[0].casefold(), note[1].casefold()))
records = fold_records(characters)
for relative in sorted(indexed_directories, key=str.casefold):
records.append(directory_record(relative))
for title, relative, content in notes:
records.append(note_record(title, relative))
records.extend(content_records(relative, content))
return records
def write_chunks(cache: Path, records: list[dict[str, Any]]) -> list[str]:
encoded_records = [encoded(record) for record in records]
chunks: list[list[str]] = []
current: list[str] = []
current_bytes = 2
for record in encoded_records:
record_bytes = len(record.encode("utf-8"))
if record_bytes > RECORD_BYTES:
raise ValueError("index record exceeded its bounded size")
separator_bytes = 1 if current else 0
if current and current_bytes + separator_bytes + record_bytes > CHUNK_BYTES:
chunks.append(current)
current = []
current_bytes = 2
separator_bytes = 0
current.append(record)
current_bytes += separator_bytes + record_bytes
if current or not chunks:
chunks.append(current)
names: list[str] = []
for index, chunk in enumerate(chunks):
chunk_name = f"index-cache-{index:03d}.json"
path = (cache.parent / chunk_name).resolve()
if path.parent != cache.parent:
raise ValueError("cache chunk escaped the plugin directory")
with tempfile.NamedTemporaryFile(
mode="w", encoding="utf-8", dir=cache.parent, delete=False
) as handle:
handle.write("[" + ",".join(chunk) + "]")
temporary = Path(handle.name)
os.replace(temporary, path)
names.append(chunk_name)
active = set(names)
for stale in cache.parent.glob(f"{cache.stem}-*.json"):
if stale.name not in active:
stale.unlink(missing_ok=True)
cache.unlink(missing_ok=True)
return names
def main() -> int:
if len(sys.argv) != 3:
print("usage: index.py VAULT CACHE", file=sys.stderr)
return 2
root = Path(sys.argv[1]).expanduser().resolve()
requested_cache = Path(sys.argv[2]).expanduser().resolve()
plugin_dir = Path(__file__).resolve().parent
if (
requested_cache.parent != plugin_dir
or requested_cache.name != "index-cache.json"
):
print("cache must be index-cache.json in the plugin directory", file=sys.stderr)
return 2
if not root.is_dir():
print(f"vault does not exist: {root}", file=sys.stderr)
return 1
records = index_records(root)
chunk_names = write_chunks(plugin_dir / "index-cache.json", records)
note_count = sum(record["kind"] == "note" for record in records)
json.dump(
{"chunks": chunk_names, "notes": note_count},
sys.stdout,
separators=(",", ":"),
)
return 0
if __name__ == "__main__":
raise SystemExit(main())