ed063ab1fe
Capture desktop-specific system, session, Noctalia, and Pi integration alongside shared laptop support. Retire obsolete X11 configuration and make package ownership explicit for repeatable Decman deployments.
251 lines
8.5 KiB
Python
251 lines
8.5 KiB
Python
#!/usr/bin/env python3
|
|
"""Build bounded JSON cache chunks for a Markdown vault."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import tempfile
|
|
import unicodedata
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
SKIP_DIRS = {
|
|
".git",
|
|
".obsidian",
|
|
".trash",
|
|
".stfolder",
|
|
"node_modules",
|
|
"dist",
|
|
"build",
|
|
}
|
|
HEADING = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE)
|
|
WORDS = re.compile(r"[^\W_]+", re.UNICODE)
|
|
CHUNK_BYTES = 96 * 1024
|
|
RECORD_BYTES = 48 * 1024
|
|
|
|
|
|
def visible_directory(name: str) -> bool:
|
|
return name not in SKIP_DIRS and not name.startswith(".")
|
|
|
|
|
|
def note_title(path: Path, content: str) -> str:
|
|
match = HEADING.search(content)
|
|
return match.group(1).strip() if match else path.stem
|
|
|
|
|
|
def folded(value: str) -> str:
|
|
"""Case-fold and strip accents so ASCII queries match Unicode notes."""
|
|
normalized = unicodedata.normalize("NFKD", value.casefold())
|
|
return "".join(
|
|
character for character in normalized if not unicodedata.combining(character)
|
|
)
|
|
|
|
|
|
def token_buckets(content: str) -> dict[str, dict[str, list[str]]]:
|
|
buckets: dict[str, dict[str, list[str]]] = {}
|
|
for token in sorted(set(WORDS.findall(content))):
|
|
if not token:
|
|
continue
|
|
by_length = buckets.setdefault(token[0], {})
|
|
by_length.setdefault(str(len(token)), []).append(token)
|
|
return buckets
|
|
|
|
|
|
def encoded(record: dict[str, Any]) -> str:
|
|
return json.dumps(record, ensure_ascii=False, separators=(",", ":"))
|
|
|
|
|
|
def fold_records(characters: set[str]) -> list[dict[str, Any]]:
|
|
mapping: dict[str, str] = {}
|
|
for character in characters:
|
|
variants = {character, character.lower(), character.upper(), character.title()}
|
|
for variant in variants:
|
|
for variant_character in variant:
|
|
mapping[variant_character] = folded(variant_character)
|
|
records: list[dict[str, Any]] = []
|
|
current: dict[str, str] = {}
|
|
for source, target in sorted(mapping.items()):
|
|
candidate = {**current, source: target}
|
|
record = {"kind": "fold", "map": candidate}
|
|
if current and len(encoded(record).encode("utf-8")) > RECORD_BYTES:
|
|
records.append({"kind": "fold", "map": current})
|
|
current = {source: target}
|
|
else:
|
|
current = candidate
|
|
if current:
|
|
records.append({"kind": "fold", "map": current})
|
|
return records
|
|
|
|
|
|
def directory_record(relative: str) -> dict[str, Any]:
|
|
record = {"kind": "directory", "relative": relative}
|
|
if len(encoded(record).encode("utf-8")) > RECORD_BYTES:
|
|
raise ValueError(f"directory metadata exceeded bounded size: {relative}")
|
|
return record
|
|
|
|
|
|
def note_record(title: str, relative: str) -> dict[str, Any]:
|
|
display_title = title[:256] + ("…" if len(title) > 256 else "")
|
|
record = {
|
|
"kind": "note",
|
|
"id": relative,
|
|
"title": display_title,
|
|
"relative": relative,
|
|
"titleRaw": display_title.casefold(),
|
|
"pathRaw": relative.casefold(),
|
|
"titleSearch": folded(display_title),
|
|
"pathSearch": folded(relative),
|
|
}
|
|
if len(encoded(record).encode("utf-8")) > RECORD_BYTES:
|
|
raise ValueError(f"note metadata exceeded bounded size: {relative}")
|
|
return record
|
|
|
|
|
|
def content_record(relative: str, fragment: str) -> dict[str, Any]:
|
|
content_search = folded(fragment)
|
|
return {
|
|
"kind": "content",
|
|
"id": relative,
|
|
"fragment": fragment,
|
|
"lines": fragment.split("\n"),
|
|
"contentRaw": fragment.casefold(),
|
|
"contentSearch": content_search,
|
|
"contentTokens": token_buckets(content_search),
|
|
}
|
|
|
|
|
|
def content_records(relative: str, content: str) -> list[dict[str, Any]]:
|
|
"""Split one note into bounded records without changing its original text."""
|
|
pending = [content]
|
|
records: list[dict[str, Any]] = []
|
|
while pending:
|
|
fragment = pending.pop()
|
|
record = content_record(relative, fragment)
|
|
if len(encoded(record).encode("utf-8")) <= RECORD_BYTES:
|
|
records.append(record)
|
|
continue
|
|
midpoint = max(len(fragment) // 2, 1)
|
|
if midpoint >= len(fragment):
|
|
raise ValueError(f"cannot split oversized content in {relative}")
|
|
pending.extend((fragment[midpoint:], fragment[:midpoint]))
|
|
return records
|
|
|
|
|
|
def index_records(root: Path) -> list[dict[str, Any]]:
|
|
notes: list[tuple[str, str, str]] = []
|
|
indexed_directories: set[str] = set()
|
|
characters: set[str] = set()
|
|
for current, directories, filenames in os.walk(root):
|
|
directories[:] = sorted(
|
|
name
|
|
for name in directories
|
|
if visible_directory(name) and not Path(current, name).is_symlink()
|
|
)
|
|
for directory in directories:
|
|
relative_directory = Path(current, directory).relative_to(root).as_posix()
|
|
indexed_directories.add(relative_directory)
|
|
characters.update(relative_directory)
|
|
for filename in sorted(filenames):
|
|
if not filename.lower().endswith(".md") or filename.startswith("."):
|
|
continue
|
|
path = Path(current, filename)
|
|
if path.is_symlink():
|
|
continue
|
|
try:
|
|
content = path.read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
continue
|
|
relative = path.relative_to(root).as_posix()
|
|
title = note_title(path, content)
|
|
characters.update(title)
|
|
characters.update(relative)
|
|
characters.update(content)
|
|
notes.append((title, relative, content))
|
|
|
|
notes.sort(key=lambda note: (note[0].casefold(), note[1].casefold()))
|
|
records = fold_records(characters)
|
|
for relative in sorted(indexed_directories, key=str.casefold):
|
|
records.append(directory_record(relative))
|
|
for title, relative, content in notes:
|
|
records.append(note_record(title, relative))
|
|
records.extend(content_records(relative, content))
|
|
return records
|
|
|
|
|
|
def write_chunks(cache: Path, records: list[dict[str, Any]]) -> list[str]:
|
|
encoded_records = [encoded(record) for record in records]
|
|
chunks: list[list[str]] = []
|
|
current: list[str] = []
|
|
current_bytes = 2
|
|
for record in encoded_records:
|
|
record_bytes = len(record.encode("utf-8"))
|
|
if record_bytes > RECORD_BYTES:
|
|
raise ValueError("index record exceeded its bounded size")
|
|
separator_bytes = 1 if current else 0
|
|
if current and current_bytes + separator_bytes + record_bytes > CHUNK_BYTES:
|
|
chunks.append(current)
|
|
current = []
|
|
current_bytes = 2
|
|
separator_bytes = 0
|
|
current.append(record)
|
|
current_bytes += separator_bytes + record_bytes
|
|
if current or not chunks:
|
|
chunks.append(current)
|
|
|
|
names: list[str] = []
|
|
for index, chunk in enumerate(chunks):
|
|
chunk_name = f"index-cache-{index:03d}.json"
|
|
path = (cache.parent / chunk_name).resolve()
|
|
if path.parent != cache.parent:
|
|
raise ValueError("cache chunk escaped the plugin directory")
|
|
with tempfile.NamedTemporaryFile(
|
|
mode="w", encoding="utf-8", dir=cache.parent, delete=False
|
|
) as handle:
|
|
handle.write("[" + ",".join(chunk) + "]")
|
|
temporary = Path(handle.name)
|
|
os.replace(temporary, path)
|
|
names.append(chunk_name)
|
|
|
|
active = set(names)
|
|
for stale in cache.parent.glob(f"{cache.stem}-*.json"):
|
|
if stale.name not in active:
|
|
stale.unlink(missing_ok=True)
|
|
cache.unlink(missing_ok=True)
|
|
return names
|
|
|
|
|
|
def main() -> int:
|
|
if len(sys.argv) != 3:
|
|
print("usage: index.py VAULT CACHE", file=sys.stderr)
|
|
return 2
|
|
|
|
root = Path(sys.argv[1]).expanduser().resolve()
|
|
requested_cache = Path(sys.argv[2]).expanduser().resolve()
|
|
plugin_dir = Path(__file__).resolve().parent
|
|
if (
|
|
requested_cache.parent != plugin_dir
|
|
or requested_cache.name != "index-cache.json"
|
|
):
|
|
print("cache must be index-cache.json in the plugin directory", file=sys.stderr)
|
|
return 2
|
|
if not root.is_dir():
|
|
print(f"vault does not exist: {root}", file=sys.stderr)
|
|
return 1
|
|
|
|
records = index_records(root)
|
|
chunk_names = write_chunks(plugin_dir / "index-cache.json", records)
|
|
note_count = sum(record["kind"] == "note" for record in records)
|
|
json.dump(
|
|
{"chunks": chunk_names, "notes": note_count},
|
|
sys.stdout,
|
|
separators=(",", ":"),
|
|
)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|