sync_scientia_glossary.py 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385
  1. #!/usr/bin/env python3
  2. """Fetch and validate an immutable machine-readable Scientia glossary snapshot."""
  3. from __future__ import annotations
  4. import argparse
  5. import hashlib
  6. import html
  7. import json
  8. import re
  9. import sys
  10. import urllib.request
  11. from collections import Counter
  12. from dataclasses import dataclass, asdict
  13. from datetime import datetime, timezone
  14. from html.parser import HTMLParser
  15. from pathlib import Path
  16. DEFAULT_URL = "https://wiki.scientia.ru/ru/glossary"
  17. SKILL_ROOT = Path(__file__).resolve().parent.parent
  18. OUTPUT_ROOT = SKILL_ROOT / "references" / "terminology" / "raw" / "scientia"
  19. def normalize_space(value: str) -> str:
  20. return re.sub(r"\s+", " ", html.unescape(value)).strip()
  21. def normalize_key(value: str) -> str:
  22. return normalize_space(value).casefold()
  23. @dataclass(frozen=True)
  24. class Entry:
  25. source_entry_id: str
  26. source_revision: str
  27. source_order: int
  28. section: str
  29. source_section: str
  30. en_raw: str
  31. ru_raw: str
  32. target_url: str
  33. article_status: str
  34. row_hash: str
  35. class GlossaryParser(HTMLParser):
  36. def __init__(self) -> None:
  37. super().__init__(convert_charrefs=True)
  38. self.section = "other"
  39. self.in_table = False
  40. self.in_row = False
  41. self.in_cell = False
  42. self.cells: list[str] = []
  43. self.cell_parts: list[str] = []
  44. self.link_href = ""
  45. self.link_status = "none"
  46. self.row_href = ""
  47. self.row_status = "none"
  48. self.rows: list[tuple[str, str, str, str, str]] = []
  49. def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
  50. attrs_map = {key: value or "" for key, value in attrs}
  51. if tag == "h1" and re.fullmatch(r"[a-z]", attrs_map.get("id", ""), re.I):
  52. self.section = attrs_map["id"].lower()
  53. elif tag == "table":
  54. self.in_table = True
  55. elif tag == "tr" and self.in_table:
  56. self.in_row = True
  57. self.cells = []
  58. self.row_href = ""
  59. self.row_status = "none"
  60. elif tag == "td" and self.in_row:
  61. self.in_cell = True
  62. self.cell_parts = []
  63. self.link_href = ""
  64. self.link_status = "none"
  65. elif tag == "a" and self.in_cell:
  66. self.link_href = attrs_map.get("href", "")
  67. classes = set(attrs_map.get("class", "").split())
  68. if "is-valid-page" in classes:
  69. self.link_status = "valid"
  70. elif "is-invalid-page" in classes:
  71. self.link_status = "missing"
  72. def handle_endtag(self, tag: str) -> None:
  73. if tag == "td" and self.in_cell:
  74. self.cells.append(normalize_space("".join(self.cell_parts)))
  75. if len(self.cells) == 2:
  76. self.row_href = self.link_href
  77. self.row_status = self.link_status
  78. self.in_cell = False
  79. elif tag == "tr" and self.in_row:
  80. if len(self.cells) == 2 and self.cells[0] and self.cells[1]:
  81. self.rows.append(
  82. (self.section, self.cells[0], self.cells[1], self.row_href, self.row_status)
  83. )
  84. self.in_row = False
  85. elif tag == "table":
  86. self.in_table = False
  87. def handle_data(self, data: str) -> None:
  88. if self.in_cell:
  89. self.cell_parts.append(data)
  90. def fetch_html(url: str) -> bytes:
  91. request = urllib.request.Request(
  92. url,
  93. headers={"User-Agent": "engineering-writing-ru glossary snapshot/0.2"},
  94. )
  95. with urllib.request.urlopen(request, timeout=30) as response:
  96. return response.read()
  97. def parse_entries(payload: bytes, revision: str) -> list[Entry]:
  98. parser = GlossaryParser()
  99. parser.feed(payload.decode("utf-8"))
  100. seen: dict[str, str] = {}
  101. entries: list[Entry] = []
  102. for order, (source_section, en_raw, ru_raw, href, status) in enumerate(parser.rows, start=1):
  103. key = normalize_key(en_raw)
  104. if key in seen:
  105. raise ValueError(f"Duplicate English key: {en_raw!r} and {seen[key]!r}")
  106. seen[key] = en_raw
  107. entry_id = "scientia-" + hashlib.sha256(key.encode("utf-8")).hexdigest()[:12]
  108. row_material = "\n".join([en_raw, ru_raw, href, status])
  109. row_hash = hashlib.sha256(row_material.encode("utf-8")).hexdigest()
  110. first_letter = re.sub(r"^[^a-z]+", "", key)
  111. section = first_letter[0] if first_letter else "other"
  112. entries.append(
  113. Entry(
  114. source_entry_id=entry_id,
  115. source_revision=revision,
  116. source_order=order,
  117. section=section,
  118. source_section=source_section,
  119. en_raw=en_raw,
  120. ru_raw=ru_raw,
  121. target_url=href,
  122. article_status=status,
  123. row_hash=row_hash,
  124. )
  125. )
  126. return entries
  127. def validate_output_root() -> None:
  128. resolved_skill = SKILL_ROOT.resolve()
  129. resolved_output = OUTPUT_ROOT.resolve()
  130. if not resolved_output.is_relative_to(resolved_skill):
  131. raise RuntimeError(f"Refusing to write outside skill root: {resolved_output}")
  132. def validate_existing() -> list[Entry]:
  133. manifest_path = OUTPUT_ROOT / "manifest.json"
  134. manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
  135. snapshot_path = OUTPUT_ROOT / manifest["raw_file"]
  136. snapshot_text = snapshot_path.read_text(encoding="utf-8")
  137. if hashlib.sha256(snapshot_text.encode("utf-8")).hexdigest() != manifest["snapshot_sha256"]:
  138. raise ValueError("Existing snapshot hash does not match manifest")
  139. entries = [
  140. Entry(**json.loads(line))
  141. for line in snapshot_text.splitlines()
  142. if line.strip()
  143. ]
  144. if len(entries) != manifest["entry_count"]:
  145. raise ValueError("Existing snapshot count does not match manifest")
  146. if len({entry.source_entry_id for entry in entries}) != len(entries):
  147. raise ValueError("Duplicate source_entry_id in existing snapshot")
  148. if len({normalize_key(entry.en_raw) for entry in entries}) != len(entries):
  149. raise ValueError("Duplicate English key in existing snapshot")
  150. for entry in entries:
  151. row_material = "\n".join(
  152. [entry.en_raw, entry.ru_raw, entry.target_url, entry.article_status]
  153. )
  154. expected_hash = hashlib.sha256(row_material.encode("utf-8")).hexdigest()
  155. if expected_hash != entry.row_hash:
  156. raise ValueError(f"Row hash mismatch: {entry.source_entry_id}")
  157. lookup_script = SKILL_ROOT / "scripts" / "lookup_scientia.py"
  158. if not lookup_script.is_file():
  159. raise ValueError("Bundled lookup_scientia.py is missing")
  160. return entries
  161. def compare_refreshed(payload: bytes) -> list[Entry]:
  162. """Compare the live table and page bytes with the immutable active snapshot."""
  163. local_entries = validate_existing()
  164. manifest = json.loads((OUTPUT_ROOT / "manifest.json").read_text(encoding="utf-8"))
  165. refreshed_entries = parse_entries(payload, str(manifest["source_revision"]))
  166. local_by_id = {entry.source_entry_id: entry for entry in local_entries}
  167. refreshed_by_id = {entry.source_entry_id: entry for entry in refreshed_entries}
  168. added = sorted(refreshed_by_id.keys() - local_by_id.keys())
  169. removed = sorted(local_by_id.keys() - refreshed_by_id.keys())
  170. def semantic_row(entry: Entry) -> tuple[object, ...]:
  171. return (
  172. entry.source_order,
  173. entry.section,
  174. entry.source_section,
  175. entry.en_raw,
  176. entry.ru_raw,
  177. entry.target_url,
  178. entry.article_status,
  179. )
  180. changed = sorted(
  181. entry_id
  182. for entry_id in local_by_id.keys() & refreshed_by_id.keys()
  183. if semantic_row(local_by_id[entry_id]) != semantic_row(refreshed_by_id[entry_id])
  184. )
  185. html_sha256 = hashlib.sha256(payload).hexdigest()
  186. html_changed = html_sha256 != manifest["html_sha256"]
  187. if added or removed or changed or html_changed:
  188. raise ValueError(
  189. "Scientia source drift detected: "
  190. f"added={len(added)}, removed={len(removed)}, changed={len(changed)}, "
  191. f"html_changed={str(html_changed).lower()}"
  192. )
  193. return refreshed_entries
  194. def write_snapshot(entries: list[Entry], payload: bytes, source_url: str, revision: str) -> None:
  195. validate_output_root()
  196. OUTPUT_ROOT.mkdir(parents=True, exist_ok=True)
  197. snapshots_root = OUTPUT_ROOT / "snapshots"
  198. snapshots_root.mkdir(parents=True, exist_ok=True)
  199. diffs_root = OUTPUT_ROOT / "diffs"
  200. diffs_root.mkdir(parents=True, exist_ok=True)
  201. snapshot_text = "\n".join(
  202. json.dumps(asdict(entry), ensure_ascii=False, sort_keys=True) for entry in entries
  203. ) + "\n"
  204. safe_revision = re.sub(r"[^0-9A-Za-z._-]+", "-", revision).strip("-")
  205. if not safe_revision:
  206. raise ValueError("source revision cannot produce an empty filename")
  207. snapshot_path = snapshots_root / f"scientia-{safe_revision}.jsonl"
  208. previous_snapshots = sorted(
  209. path for path in snapshots_root.glob("scientia-*.jsonl") if path != snapshot_path
  210. )
  211. if snapshot_path.exists():
  212. existing = snapshot_path.read_text(encoding="utf-8")
  213. if existing != snapshot_text:
  214. raise RuntimeError(
  215. f"Immutable snapshot already exists with different content: {snapshot_path.name}"
  216. )
  217. else:
  218. snapshot_path.write_text(snapshot_text, encoding="utf-8", newline="\n")
  219. if previous_snapshots:
  220. previous_path = previous_snapshots[-1]
  221. previous_entries = {
  222. item["source_entry_id"]: item
  223. for item in (
  224. json.loads(line)
  225. for line in previous_path.read_text(encoding="utf-8").splitlines()
  226. if line.strip()
  227. )
  228. }
  229. current_entries = {entry.source_entry_id: asdict(entry) for entry in entries}
  230. added = sorted(current_entries.keys() - previous_entries.keys())
  231. removed = sorted(previous_entries.keys() - current_entries.keys())
  232. changed = sorted(
  233. key
  234. for key in current_entries.keys() & previous_entries.keys()
  235. if current_entries[key]["row_hash"] != previous_entries[key]["row_hash"]
  236. )
  237. diff = {
  238. "from": previous_path.name,
  239. "to": snapshot_path.name,
  240. "added": added,
  241. "removed": removed,
  242. "changed": changed,
  243. "requires_override_review": bool(removed or changed),
  244. }
  245. diff_name = f"{previous_path.stem}--{snapshot_path.stem}.json"
  246. (diffs_root / diff_name).write_text(
  247. json.dumps(diff, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
  248. encoding="utf-8",
  249. newline="\n",
  250. )
  251. sections = Counter(
  252. entry.section if re.fullmatch(r"[a-z]", entry.section) else "other"
  253. for entry in entries
  254. )
  255. manifest = {
  256. "schema_version": 2,
  257. "source_url": source_url,
  258. "source_revision": revision,
  259. "captured_at_utc": datetime.now(timezone.utc).replace(microsecond=0).isoformat(),
  260. "entry_count": len(entries),
  261. "html_sha256": hashlib.sha256(payload).hexdigest(),
  262. "snapshot_sha256": hashlib.sha256(snapshot_text.encode("utf-8")).hexdigest(),
  263. "sections": dict(sorted(sections.items())),
  264. "raw_file": snapshot_path.relative_to(OUTPUT_ROOT).as_posix(),
  265. "runtime_lookup": "../../../../scripts/lookup_scientia.py",
  266. "lookup_fields": ["en_raw", "ru_raw"],
  267. }
  268. (OUTPUT_ROOT / "manifest.json").write_text(
  269. json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
  270. encoding="utf-8",
  271. newline="\n",
  272. )
  273. def build_parser() -> argparse.ArgumentParser:
  274. parser = argparse.ArgumentParser()
  275. parser.add_argument("--url", default=DEFAULT_URL)
  276. parser.add_argument("--input-html", type=Path)
  277. parser.add_argument("--source-revision", default="2025-09-12")
  278. parser.add_argument("--expected-count", type=int)
  279. parser.add_argument("--check-only", action="store_true")
  280. parser.add_argument(
  281. "--refresh",
  282. action="store_true",
  283. help="Fetch the source even in check-only mode instead of validating the local snapshot.",
  284. )
  285. return parser
  286. def main() -> int:
  287. args = build_parser().parse_args()
  288. if args.check_only and not args.refresh and args.input_html is None:
  289. entries = validate_existing()
  290. if args.expected_count is not None and len(entries) != args.expected_count:
  291. raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}")
  292. print(
  293. json.dumps(
  294. {
  295. "entries": len(entries),
  296. "sections": sorted({entry.section for entry in entries}),
  297. "check_only": True,
  298. "source": "local-snapshot",
  299. },
  300. ensure_ascii=False,
  301. )
  302. )
  303. return 0
  304. payload = args.input_html.read_bytes() if args.input_html else fetch_html(args.url)
  305. if args.check_only:
  306. entries = compare_refreshed(payload)
  307. if args.expected_count is not None and len(entries) != args.expected_count:
  308. raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}")
  309. print(
  310. json.dumps(
  311. {
  312. "entries": len(entries),
  313. "sections": sorted({entry.section for entry in entries}),
  314. "check_only": True,
  315. "source": "refreshed-source",
  316. "matches_active_snapshot": True,
  317. },
  318. ensure_ascii=False,
  319. )
  320. )
  321. return 0
  322. entries = parse_entries(payload, args.source_revision)
  323. if args.expected_count is not None and len(entries) != args.expected_count:
  324. raise ValueError(f"Expected {args.expected_count} entries, found {len(entries)}")
  325. write_snapshot(entries, payload, args.url, args.source_revision)
  326. print(
  327. json.dumps(
  328. {
  329. "entries": len(entries),
  330. "sections": sorted({entry.section for entry in entries}),
  331. "check_only": args.check_only,
  332. },
  333. ensure_ascii=False,
  334. )
  335. )
  336. return 0
  337. if __name__ == "__main__":
  338. try:
  339. raise SystemExit(main())
  340. except Exception as exc:
  341. print(f"ERROR: {exc}", file=sys.stderr)
  342. raise