| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385 |
- #!/usr/bin/env python3
- """Fetch and validate an immutable machine-readable Scientia glossary snapshot."""
- from __future__ import annotations
- import argparse
- import hashlib
- import html
- import json
- import re
- import sys
- import urllib.request
- from collections import Counter
- from dataclasses import dataclass, asdict
- from datetime import datetime, timezone
- from html.parser import HTMLParser
- from pathlib import Path
- DEFAULT_URL = "https://wiki.scientia.ru/ru/glossary"
- SKILL_ROOT = Path(__file__).resolve().parent.parent
- OUTPUT_ROOT = SKILL_ROOT / "references" / "terminology" / "raw" / "scientia"
- def normalize_space(value: str) -> str:
- return re.sub(r"\s+", " ", html.unescape(value)).strip()
- def normalize_key(value: str) -> str:
- return normalize_space(value).casefold()
- @dataclass(frozen=True)
- class Entry:
- source_entry_id: str
- source_revision: str
- source_order: int
- section: str
- source_section: str
- en_raw: str
- ru_raw: str
- target_url: str
- article_status: str
- row_hash: str
- class GlossaryParser(HTMLParser):
- def __init__(self) -> None:
- super().__init__(convert_charrefs=True)
- self.section = "other"
- self.in_table = False
- self.in_row = False
- self.in_cell = False
- self.cells: list[str] = []
- self.cell_parts: list[str] = []
- self.link_href = ""
- self.link_status = "none"
- self.row_href = ""
- self.row_status = "none"
- self.rows: list[tuple[str, str, str, str, str]] = []
- def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
- attrs_map = {key: value or "" for key, value in attrs}
- if tag == "h1" and re.fullmatch(r"[a-z]", attrs_map.get("id", ""), re.I):
- self.section = attrs_map["id"].lower()
- elif tag == "table":
- self.in_table = True
- elif tag == "tr" and self.in_table:
- self.in_row = True
- self.cells = []
- self.row_href = ""
- self.row_status = "none"
- elif tag == "td" and self.in_row:
- self.in_cell = True
- self.cell_parts = []
- self.link_href = ""
- self.link_status = "none"
- elif tag == "a" and self.in_cell:
- self.link_href = attrs_map.get("href", "")
- classes = set(attrs_map.get("class", "").split())
- if "is-valid-page" in classes:
- self.link_status = "valid"
- elif "is-invalid-page" in classes:
- self.link_status = "missing"
- def handle_endtag(self, tag: str) -> None:
- if tag == "td" and self.in_cell:
- self.cells.append(normalize_space("".join(self.cell_parts)))
- if len(self.cells) == 2:
- self.row_href = self.link_href
- self.row_status = self.link_status
- self.in_cell = False
- elif tag == "tr" and self.in_row:
- if len(self.cells) == 2 and self.cells[0] and self.cells[1]:
- self.rows.append(
- (self.section, self.cells[0], self.cells[1], self.row_href, self.row_status)
- )
- self.in_row = False
- elif tag == "table":
- self.in_table = False
- def handle_data(self, data: str) -> None:
- if self.in_cell:
- self.cell_parts.append(data)
- def fetch_html(url: str) -> bytes:
- request = urllib.request.Request(
- url,
- headers={"User-Agent": "engineering-writing-ru glossary snapshot/0.2"},
- )
- with urllib.request.urlopen(request, timeout=30) as response:
- return response.read()
- def parse_entries(payload: bytes, revision: str) -> list[Entry]:
- parser = GlossaryParser()
- parser.feed(payload.decode("utf-8"))
- seen: dict[str, str] = {}
- entries: list[Entry] = []
- for order, (source_section, en_raw, ru_raw, href, status) in enumerate(parser.rows, start=1):
- key = normalize_key(en_raw)
- if key in seen:
- raise ValueError(f"Duplicate English key: {en_raw!r} and {seen[key]!r}")
- seen[key] = en_raw
- entry_id = "scientia-" + hashlib.sha256(key.encode("utf-8")).hexdigest()[:12]
- row_material = "\n".join([en_raw, ru_raw, href, status])
- row_hash = hashlib.sha256(row_material.encode("utf-8")).hexdigest()
- first_letter = re.sub(r"^[^a-z]+", "", key)
- section = first_letter[0] if first_letter else "other"
- entries.append(
- Entry(
- source_entry_id=entry_id,
- source_revision=revision,
- source_order=order,
- section=section,
- source_section=source_section,
- en_raw=en_raw,
- ru_raw=ru_raw,
- target_url=href,
- article_status=status,
- row_hash=row_hash,
- )
- )
- return entries
- def validate_output_root() -> None:
- resolved_skill = SKILL_ROOT.resolve()
- resolved_output = OUTPUT_ROOT.resolve()
- if not resolved_output.is_relative_to(resolved_skill):
- raise RuntimeError(f"Refusing to write outside skill root: {resolved_output}")
- def validate_existing() -> list[Entry]:
- manifest_path = OUTPUT_ROOT / "manifest.json"
- manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
- snapshot_path = OUTPUT_ROOT / manifest["raw_file"]
- snapshot_text = snapshot_path.read_text(encoding="utf-8")
- if hashlib.sha256(snapshot_text.encode("utf-8")).hexdigest() != manifest["snapshot_sha256"]:
- raise ValueError("Existing snapshot hash does not match manifest")
- entries = [
- Entry(**json.loads(line))
- for line in snapshot_text.splitlines()
- if line.strip()
- ]
- if len(entries) != manifest["entry_count"]:
- raise ValueError("Existing snapshot count does not match manifest")
- if len({entry.source_entry_id for entry in entries}) != len(entries):
- raise ValueError("Duplicate source_entry_id in existing snapshot")
- if len({normalize_key(entry.en_raw) for entry in entries}) != len(entries):
- raise ValueError("Duplicate English key in existing snapshot")
- for entry in entries:
- row_material = "\n".join(
- [entry.en_raw, entry.ru_raw, entry.target_url, entry.article_status]
- )
- expected_hash = hashlib.sha256(row_material.encode("utf-8")).hexdigest()
- if expected_hash != entry.row_hash:
- raise ValueError(f"Row hash mismatch: {entry.source_entry_id}")
- lookup_script = SKILL_ROOT / "scripts" / "lookup_scientia.py"
- if not lookup_script.is_file():
- raise ValueError("Bundled lookup_scientia.py is missing")
- return entries
- def compare_refreshed(payload: bytes) -> list[Entry]:
- """Compare the live table and page bytes with the immutable active snapshot."""
- local_entries = validate_existing()
- manifest = json.loads((OUTPUT_ROOT / "manifest.json").read_text(encoding="utf-8"))
- refreshed_entries = parse_entries(payload, str(manifest["source_revision"]))
- local_by_id = {entry.source_entry_id: entry for entry in local_entries}
- refreshed_by_id = {entry.source_entry_id: entry for entry in refreshed_entries}
- added = sorted(refreshed_by_id.keys() - local_by_id.keys())
- removed = sorted(local_by_id.keys() - refreshed_by_id.keys())
- def semantic_row(entry: Entry) -> tuple[object, ...]:
- return (
- entry.source_order,
- entry.section,
- entry.source_section,
- entry.en_raw,
- entry.ru_raw,
- entry.target_url,
- entry.article_status,
- )
- changed = sorted(
- entry_id
- for entry_id in local_by_id.keys() & refreshed_by_id.keys()
- if semantic_row(local_by_id[entry_id]) != semantic_row(refreshed_by_id[entry_id])
- )
- html_sha256 = hashlib.sha256(payload).hexdigest()
- html_changed = html_sha256 != manifest["html_sha256"]
- if added or removed or changed or html_changed:
- raise ValueError(
- "Scientia source drift detected: "
- f"added={len(added)}, removed={len(removed)}, changed={len(changed)}, "
- f"html_changed={str(html_changed).lower()}"
- )
- return refreshed_entries
- def write_snapshot(entries: list[Entry], payload: bytes, source_url: str, revision: str) -> None:
- validate_output_root()
- OUTPUT_ROOT.mkdir(parents=True, exist_ok=True)
- snapshots_root = OUTPUT_ROOT / "snapshots"
- snapshots_root.mkdir(parents=True, exist_ok=True)
- diffs_root = OUTPUT_ROOT / "diffs"
- diffs_root.mkdir(parents=True, exist_ok=True)
- snapshot_text = "\n".join(
- json.dumps(asdict(entry), ensure_ascii=False, sort_keys=True) for entry in entries
- ) + "\n"
- safe_revision = re.sub(r"[^0-9A-Za-z._-]+", "-", revision).strip("-")
- if not safe_revision:
- raise ValueError("source revision cannot produce an empty filename")
- snapshot_path = snapshots_root / f"scientia-{safe_revision}.jsonl"
- previous_snapshots = sorted(
- path for path in snapshots_root.glob("scientia-*.jsonl") if path != snapshot_path
- )
- if snapshot_path.exists():
- existing = snapshot_path.read_text(encoding="utf-8")
- if existing != snapshot_text:
- raise RuntimeError(
- f"Immutable snapshot already exists with different content: {snapshot_path.name}"
- )
- else:
- snapshot_path.write_text(snapshot_text, encoding="utf-8", newline="\n")
- if previous_snapshots:
- previous_path = previous_snapshots[-1]
- previous_entries = {
- item["source_entry_id"]: item
- for item in (
- json.loads(line)
- for line in previous_path.read_text(encoding="utf-8").splitlines()
- if line.strip()
- )
- }
- current_entries = {entry.source_entry_id: asdict(entry) for entry in entries}
- added = sorted(current_entries.keys() - previous_entries.keys())
- removed = sorted(previous_entries.keys() - current_entries.keys())
- changed = sorted(
- key
- for key in current_entries.keys() & previous_entries.keys()
- if current_entries[key]["row_hash"] != previous_entries[key]["row_hash"]
- )
- diff = {
- "from": previous_path.name,
- "to": snapshot_path.name,
- "added": added,
- "removed": removed,
- "changed": changed,
- "requires_override_review": bool(removed or changed),
- }
- diff_name = f"{previous_path.stem}--{snapshot_path.stem}.json"
- (diffs_root / diff_name).write_text(
- json.dumps(diff, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
- encoding="utf-8",
- newline="\n",
- )
- sections = Counter(
- entry.section if re.fullmatch(r"[a-z]", entry.section) else "other"
- for entry in entries
- )
- manifest = {
- "schema_version": 2,
- "source_url": source_url,
- "source_revision": revision,
- "captured_at_utc": datetime.now(timezone.utc).replace(microsecond=0).isoformat(),
- "entry_count": len(entries),
- "html_sha256": hashlib.sha256(payload).hexdigest(),
- "snapshot_sha256": hashlib.sha256(snapshot_text.encode("utf-8")).hexdigest(),
- "sections": dict(sorted(sections.items())),
- "raw_file": snapshot_path.relative_to(OUTPUT_ROOT).as_posix(),
- "runtime_lookup": "../../../../scripts/lookup_scientia.py",
- "lookup_fields": ["en_raw", "ru_raw"],
- }
- (OUTPUT_ROOT / "manifest.json").write_text(
- json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
- encoding="utf-8",
- newline="\n",
- )
- def build_parser() -> argparse.ArgumentParser:
- parser = argparse.ArgumentParser()
- parser.add_argument("--url", default=DEFAULT_URL)
- parser.add_argument("--input-html", type=Path)
- parser.add_argument("--source-revision", default="2025-09-12")
- parser.add_argument("--expected-count", type=int)
- parser.add_argument("--check-only", action="store_true")
- parser.add_argument(
- "--refresh",
- action="store_true",
- help="Fetch the source even in check-only mode instead of validating the local snapshot.",
- )
- return parser
- def main() -> int:
- args = build_parser().parse_args()
- if args.check_only and not args.refresh and args.input_html is None:
- entries = validate_existing()
- if args.expected_count is not None and len(entries) != args.expected_count:
- raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}")
- print(
- json.dumps(
- {
- "entries": len(entries),
- "sections": sorted({entry.section for entry in entries}),
- "check_only": True,
- "source": "local-snapshot",
- },
- ensure_ascii=False,
- )
- )
- return 0
- payload = args.input_html.read_bytes() if args.input_html else fetch_html(args.url)
- if args.check_only:
- entries = compare_refreshed(payload)
- if args.expected_count is not None and len(entries) != args.expected_count:
- raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}")
- print(
- json.dumps(
- {
- "entries": len(entries),
- "sections": sorted({entry.section for entry in entries}),
- "check_only": True,
- "source": "refreshed-source",
- "matches_active_snapshot": True,
- },
- ensure_ascii=False,
- )
- )
- return 0
- entries = parse_entries(payload, args.source_revision)
- if args.expected_count is not None and len(entries) != args.expected_count:
- raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}")
- write_snapshot(entries, payload, args.url, args.source_revision)
- print(
- json.dumps(
- {
- "entries": len(entries),
- "sections": sorted({entry.section for entry in entries}),
- "check_only": args.check_only,
- },
- ensure_ascii=False,
- )
- )
- return 0
- if __name__ == "__main__":
- try:
- raise SystemExit(main())
- except Exception as exc:
- print(f"ERROR: {exc}", file=sys.stderr)
- raise
|