#!/usr/bin/env python3 """Fetch and validate an immutable machine-readable Scientia glossary snapshot.""" from __future__ import annotations import argparse import hashlib import html import json import re import sys import urllib.request from collections import Counter from dataclasses import dataclass, asdict from datetime import datetime, timezone from html.parser import HTMLParser from pathlib import Path DEFAULT_URL = "https://wiki.scientia.ru/ru/glossary" SKILL_ROOT = Path(__file__).resolve().parent.parent OUTPUT_ROOT = SKILL_ROOT / "references" / "terminology" / "raw" / "scientia" def normalize_space(value: str) -> str: return re.sub(r"\s+", " ", html.unescape(value)).strip() def normalize_key(value: str) -> str: return normalize_space(value).casefold() @dataclass(frozen=True) class Entry: source_entry_id: str source_revision: str source_order: int section: str source_section: str en_raw: str ru_raw: str target_url: str article_status: str row_hash: str class GlossaryParser(HTMLParser): def __init__(self) -> None: super().__init__(convert_charrefs=True) self.section = "other" self.in_table = False self.in_row = False self.in_cell = False self.cells: list[str] = [] self.cell_parts: list[str] = [] self.link_href = "" self.link_status = "none" self.row_href = "" self.row_status = "none" self.rows: list[tuple[str, str, str, str, str]] = [] def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: attrs_map = {key: value or "" for key, value in attrs} if tag == "h1" and re.fullmatch(r"[a-z]", attrs_map.get("id", ""), re.I): self.section = attrs_map["id"].lower() elif tag == "table": self.in_table = True elif tag == "tr" and self.in_table: self.in_row = True self.cells = [] self.row_href = "" self.row_status = "none" elif tag == "td" and self.in_row: self.in_cell = True self.cell_parts = [] self.link_href = "" self.link_status = "none" elif tag == "a" and self.in_cell: self.link_href = attrs_map.get("href", "") classes = set(attrs_map.get("class", "").split()) if "is-valid-page" in classes: self.link_status = "valid" elif "is-invalid-page" in classes: self.link_status = "missing" def handle_endtag(self, tag: str) -> None: if tag == "td" and self.in_cell: self.cells.append(normalize_space("".join(self.cell_parts))) if len(self.cells) == 2: self.row_href = self.link_href self.row_status = self.link_status self.in_cell = False elif tag == "tr" and self.in_row: if len(self.cells) == 2 and self.cells[0] and self.cells[1]: self.rows.append( (self.section, self.cells[0], self.cells[1], self.row_href, self.row_status) ) self.in_row = False elif tag == "table": self.in_table = False def handle_data(self, data: str) -> None: if self.in_cell: self.cell_parts.append(data) def fetch_html(url: str) -> bytes: request = urllib.request.Request( url, headers={"User-Agent": "engineering-writing-ru glossary snapshot/0.2"}, ) with urllib.request.urlopen(request, timeout=30) as response: return response.read() def parse_entries(payload: bytes, revision: str) -> list[Entry]: parser = GlossaryParser() parser.feed(payload.decode("utf-8")) seen: dict[str, str] = {} entries: list[Entry] = [] for order, (source_section, en_raw, ru_raw, href, status) in enumerate(parser.rows, start=1): key = normalize_key(en_raw) if key in seen: raise ValueError(f"Duplicate English key: {en_raw!r} and {seen[key]!r}") seen[key] = en_raw entry_id = "scientia-" + hashlib.sha256(key.encode("utf-8")).hexdigest()[:12] row_material = "\n".join([en_raw, ru_raw, href, status]) row_hash = hashlib.sha256(row_material.encode("utf-8")).hexdigest() first_letter = re.sub(r"^[^a-z]+", "", key) section = first_letter[0] if first_letter else "other" entries.append( Entry( source_entry_id=entry_id, source_revision=revision, source_order=order, section=section, source_section=source_section, en_raw=en_raw, ru_raw=ru_raw, target_url=href, article_status=status, row_hash=row_hash, ) ) return entries def validate_output_root() -> None: resolved_skill = SKILL_ROOT.resolve() resolved_output = OUTPUT_ROOT.resolve() if not resolved_output.is_relative_to(resolved_skill): raise RuntimeError(f"Refusing to write outside skill root: {resolved_output}") def validate_existing() -> list[Entry]: manifest_path = OUTPUT_ROOT / "manifest.json" manifest = json.loads(manifest_path.read_text(encoding="utf-8")) snapshot_path = OUTPUT_ROOT / manifest["raw_file"] snapshot_text = snapshot_path.read_text(encoding="utf-8") if hashlib.sha256(snapshot_text.encode("utf-8")).hexdigest() != manifest["snapshot_sha256"]: raise ValueError("Existing snapshot hash does not match manifest") entries = [ Entry(**json.loads(line)) for line in snapshot_text.splitlines() if line.strip() ] if len(entries) != manifest["entry_count"]: raise ValueError("Existing snapshot count does not match manifest") if len({entry.source_entry_id for entry in entries}) != len(entries): raise ValueError("Duplicate source_entry_id in existing snapshot") if len({normalize_key(entry.en_raw) for entry in entries}) != len(entries): raise ValueError("Duplicate English key in existing snapshot") for entry in entries: row_material = "\n".join( [entry.en_raw, entry.ru_raw, entry.target_url, entry.article_status] ) expected_hash = hashlib.sha256(row_material.encode("utf-8")).hexdigest() if expected_hash != entry.row_hash: raise ValueError(f"Row hash mismatch: {entry.source_entry_id}") lookup_script = SKILL_ROOT / "scripts" / "lookup_scientia.py" if not lookup_script.is_file(): raise ValueError("Bundled lookup_scientia.py is missing") return entries def compare_refreshed(payload: bytes) -> list[Entry]: """Compare the live table and page bytes with the immutable active snapshot.""" local_entries = validate_existing() manifest = json.loads((OUTPUT_ROOT / "manifest.json").read_text(encoding="utf-8")) refreshed_entries = parse_entries(payload, str(manifest["source_revision"])) local_by_id = {entry.source_entry_id: entry for entry in local_entries} refreshed_by_id = {entry.source_entry_id: entry for entry in refreshed_entries} added = sorted(refreshed_by_id.keys() - local_by_id.keys()) removed = sorted(local_by_id.keys() - refreshed_by_id.keys()) def semantic_row(entry: Entry) -> tuple[object, ...]: return ( entry.source_order, entry.section, entry.source_section, entry.en_raw, entry.ru_raw, entry.target_url, entry.article_status, ) changed = sorted( entry_id for entry_id in local_by_id.keys() & refreshed_by_id.keys() if semantic_row(local_by_id[entry_id]) != semantic_row(refreshed_by_id[entry_id]) ) html_sha256 = hashlib.sha256(payload).hexdigest() html_changed = html_sha256 != manifest["html_sha256"] if added or removed or changed or html_changed: raise ValueError( "Scientia source drift detected: " f"added={len(added)}, removed={len(removed)}, changed={len(changed)}, " f"html_changed={str(html_changed).lower()}" ) return refreshed_entries def write_snapshot(entries: list[Entry], payload: bytes, source_url: str, revision: str) -> None: validate_output_root() OUTPUT_ROOT.mkdir(parents=True, exist_ok=True) snapshots_root = OUTPUT_ROOT / "snapshots" snapshots_root.mkdir(parents=True, exist_ok=True) diffs_root = OUTPUT_ROOT / "diffs" diffs_root.mkdir(parents=True, exist_ok=True) snapshot_text = "\n".join( json.dumps(asdict(entry), ensure_ascii=False, sort_keys=True) for entry in entries ) + "\n" safe_revision = re.sub(r"[^0-9A-Za-z._-]+", "-", revision).strip("-") if not safe_revision: raise ValueError("source revision cannot produce an empty filename") snapshot_path = snapshots_root / f"scientia-{safe_revision}.jsonl" previous_snapshots = sorted( path for path in snapshots_root.glob("scientia-*.jsonl") if path != snapshot_path ) if snapshot_path.exists(): existing = snapshot_path.read_text(encoding="utf-8") if existing != snapshot_text: raise RuntimeError( f"Immutable snapshot already exists with different content: {snapshot_path.name}" ) else: snapshot_path.write_text(snapshot_text, encoding="utf-8", newline="\n") if previous_snapshots: previous_path = previous_snapshots[-1] previous_entries = { item["source_entry_id"]: item for item in ( json.loads(line) for line in previous_path.read_text(encoding="utf-8").splitlines() if line.strip() ) } current_entries = {entry.source_entry_id: asdict(entry) for entry in entries} added = sorted(current_entries.keys() - previous_entries.keys()) removed = sorted(previous_entries.keys() - current_entries.keys()) changed = sorted( key for key in current_entries.keys() & previous_entries.keys() if current_entries[key]["row_hash"] != previous_entries[key]["row_hash"] ) diff = { "from": previous_path.name, "to": snapshot_path.name, "added": added, "removed": removed, "changed": changed, "requires_override_review": bool(removed or changed), } diff_name = f"{previous_path.stem}--{snapshot_path.stem}.json" (diffs_root / diff_name).write_text( json.dumps(diff, ensure_ascii=False, indent=2, sort_keys=True) + "\n", encoding="utf-8", newline="\n", ) sections = Counter( entry.section if re.fullmatch(r"[a-z]", entry.section) else "other" for entry in entries ) manifest = { "schema_version": 2, "source_url": source_url, "source_revision": revision, "captured_at_utc": datetime.now(timezone.utc).replace(microsecond=0).isoformat(), "entry_count": len(entries), "html_sha256": hashlib.sha256(payload).hexdigest(), "snapshot_sha256": hashlib.sha256(snapshot_text.encode("utf-8")).hexdigest(), "sections": dict(sorted(sections.items())), "raw_file": snapshot_path.relative_to(OUTPUT_ROOT).as_posix(), "runtime_lookup": "../../../../scripts/lookup_scientia.py", "lookup_fields": ["en_raw", "ru_raw"], } (OUTPUT_ROOT / "manifest.json").write_text( json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True) + "\n", encoding="utf-8", newline="\n", ) def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser() parser.add_argument("--url", default=DEFAULT_URL) parser.add_argument("--input-html", type=Path) parser.add_argument("--source-revision", default="2025-09-12") parser.add_argument("--expected-count", type=int) parser.add_argument("--check-only", action="store_true") parser.add_argument( "--refresh", action="store_true", help="Fetch the source even in check-only mode instead of validating the local snapshot.", ) return parser def main() -> int: args = build_parser().parse_args() if args.check_only and not args.refresh and args.input_html is None: entries = validate_existing() if args.expected_count is not None and len(entries) != args.expected_count: raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}") print( json.dumps( { "entries": len(entries), "sections": sorted({entry.section for entry in entries}), "check_only": True, "source": "local-snapshot", }, ensure_ascii=False, ) ) return 0 payload = args.input_html.read_bytes() if args.input_html else fetch_html(args.url) if args.check_only: entries = compare_refreshed(payload) if args.expected_count is not None and len(entries) != args.expected_count: raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}") print( json.dumps( { "entries": len(entries), "sections": sorted({entry.section for entry in entries}), "check_only": True, "source": "refreshed-source", "matches_active_snapshot": True, }, ensure_ascii=False, ) ) return 0 entries = parse_entries(payload, args.source_revision) if args.expected_count is not None and len(entries) != args.expected_count: raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}") write_snapshot(entries, payload, args.url, args.source_revision) print( json.dumps( { "entries": len(entries), "sections": sorted({entry.section for entry in entries}), "check_only": args.check_only, }, ensure_ascii=False, ) ) return 0 if __name__ == "__main__": try: raise SystemExit(main()) except Exception as exc: print(f"ERROR: {exc}", file=sys.stderr) raise