lookup_scientia.py 19 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546
  1. #!/usr/bin/env python3
  2. """Select a compact, semantic terminology context without exposing raw metadata."""
  3. from __future__ import annotations
  4. import argparse
  5. import csv
  6. import hashlib
  7. import io
  8. import json
  9. import re
  10. import unicodedata
  11. from dataclasses import dataclass
  12. from pathlib import Path
  13. from typing import Iterable
  14. SKILL_ROOT = Path(__file__).resolve().parent.parent
  15. TERMINOLOGY_ROOT = SKILL_ROOT / "references" / "terminology"
  16. RAW_ROOT = TERMINOLOGY_ROOT / "raw" / "scientia"
  17. MANIFEST_PATH = RAW_ROOT / "manifest.json"
  18. APPROVED_CORE_PATH = TERMINOLOGY_ROOT / "runtime" / "approved-core.tsv"
  19. CONTEXT_PATH = TERMINOLOGY_ROOT / "runtime" / "context-dependent.tsv"
  20. PENDING_PATH = TERMINOLOGY_ROOT / "maintenance" / "pending-corrections.tsv"
  21. TOKEN_RE = re.compile(r"[0-9a-zа-я]+", re.IGNORECASE)
  22. MAX_INPUT_CHARS = 5_000_000
  23. @dataclass(frozen=True)
  24. class TermRecord:
  25. english: str
  26. russian: str
  27. condition: str
  28. aliases: tuple[str, ...]
  29. source: str
  30. priority: int
  31. raw_candidate: bool = False
  32. @dataclass(frozen=True)
  33. class Hit:
  34. start: int
  35. end: int
  36. record: TermRecord
  37. matched: str
  38. def normalized_tokens(value: str) -> tuple[str, ...]:
  39. value = unicodedata.normalize("NFKC", value).casefold().replace("ё", "е")
  40. return tuple(TOKEN_RE.findall(value))
  41. def normalized_text(value: str) -> str:
  42. return " ".join(normalized_tokens(value))
  43. def load_snapshot() -> list[dict[str, object]]:
  44. manifest = json.loads(MANIFEST_PATH.read_text(encoding="utf-8"))
  45. snapshot_path = RAW_ROOT / str(manifest["raw_file"])
  46. payload = snapshot_path.read_bytes()
  47. if hashlib.sha256(payload).hexdigest() != manifest["snapshot_sha256"]:
  48. raise ValueError("Scientia snapshot hash does not match manifest")
  49. entries = [
  50. json.loads(line)
  51. for line in payload.decode("utf-8").splitlines()
  52. if line.strip()
  53. ]
  54. if len(entries) != manifest["entry_count"]:
  55. raise ValueError("Scientia snapshot count does not match manifest")
  56. return entries
  57. def read_tsv(path: Path, required: set[str]) -> list[dict[str, str]]:
  58. lines = [
  59. line
  60. for line in path.read_text(encoding="utf-8-sig").splitlines()
  61. if line.strip() and not line.lstrip().startswith("#")
  62. ]
  63. if not lines:
  64. raise ValueError(f"TSV has no header: {path}")
  65. reader = csv.DictReader(io.StringIO("\n".join(lines)), delimiter="\t")
  66. fields = set(reader.fieldnames or [])
  67. missing = required - fields
  68. if missing:
  69. raise ValueError(f"TSV {path} is missing columns: {', '.join(sorted(missing))}")
  70. return [
  71. {key: (value or "").strip() for key, value in row.items() if key is not None}
  72. for row in reader
  73. ]
  74. def split_aliases(value: str) -> tuple[str, ...]:
  75. return tuple(item.strip() for item in value.split("|") if item.strip())
  76. def lexical_variants(value: str) -> tuple[str, ...]:
  77. """Return useful written variants without inventing translations."""
  78. variants = [value]
  79. without_parentheses = re.sub(r"\([^)]*\)", " ", value).strip()
  80. if without_parentheses:
  81. variants.append(without_parentheses)
  82. for group in re.findall(r"\(([^)]*)\)", value):
  83. variants.extend(re.split(r"[,;/=]+", group))
  84. variants.extend(re.split(r"[,;/]+", value))
  85. result: list[str] = []
  86. seen: set[tuple[str, ...]] = set()
  87. for variant in variants:
  88. variant = variant.strip(" =–—-\t")
  89. tokens = normalized_tokens(variant)
  90. if tokens and tokens not in seen:
  91. result.append(variant)
  92. seen.add(tokens)
  93. return tuple(result)
  94. def load_pending_keys(
  95. entries: list[dict[str, object]],
  96. ) -> tuple[set[str], set[str], set[tuple[str, ...]]]:
  97. rows = read_tsv(
  98. PENDING_PATH,
  99. {"english_raw", "russian_raw", "status"},
  100. )
  101. by_english = {normalized_text(str(entry["en_raw"])): entry for entry in entries}
  102. blocked_english: set[str] = set()
  103. blocked_russian: set[str] = set()
  104. pending_forms: set[tuple[str, ...]] = set()
  105. for row in rows:
  106. if row["status"] != "pending":
  107. raise ValueError("Only pending records belong in pending-corrections.tsv")
  108. key = normalized_text(row["english_raw"])
  109. source = by_english.get(key)
  110. if source is None or str(source["en_raw"]) != row["english_raw"]:
  111. raise ValueError(f"Pending record is absent from raw snapshot: {row['english_raw']}")
  112. if str(source["ru_raw"]) != row["russian_raw"]:
  113. raise ValueError(f"Pending Russian raw value differs from snapshot: {row['english_raw']}")
  114. blocked_english.add(key)
  115. blocked_russian.add(normalized_text(row["russian_raw"]))
  116. for value in (row["english_raw"], row["russian_raw"]):
  117. for variant in lexical_variants(value):
  118. tokens = normalized_tokens(variant)
  119. if tokens:
  120. pending_forms.add(tokens)
  121. return blocked_english, blocked_russian, pending_forms
  122. def load_approved_core(entries: list[dict[str, object]]) -> list[TermRecord]:
  123. rows = read_tsv(APPROVED_CORE_PATH, {"english", "russian", "sense", "aliases"})
  124. raw_pairs = {(str(item["en_raw"]), str(item["ru_raw"])) for item in entries}
  125. records: list[TermRecord] = []
  126. seen: set[str] = set()
  127. for row in rows:
  128. pair = (row["english"], row["russian"])
  129. if pair not in raw_pairs:
  130. raise ValueError(
  131. "approved-core.tsv must contain an exact positive pair from the snapshot: "
  132. f"{row['english']} — {row['russian']}"
  133. )
  134. key = normalized_text(row["english"])
  135. if not key or key in seen:
  136. raise ValueError(f"Duplicate or blank approved term: {row['english']}")
  137. seen.add(key)
  138. records.append(
  139. TermRecord(
  140. english=row["english"],
  141. russian=row["russian"],
  142. condition=row["sense"],
  143. aliases=split_aliases(row["aliases"]),
  144. source="core glossary",
  145. priority=200,
  146. )
  147. )
  148. return records
  149. def load_context_records() -> list[TermRecord]:
  150. rows = read_tsv(
  151. CONTEXT_PATH,
  152. {
  153. "term",
  154. "language",
  155. "possible_renderings",
  156. "decision_rule",
  157. "context_cues",
  158. "aliases",
  159. },
  160. )
  161. records: list[TermRecord] = []
  162. for row in rows:
  163. if row["language"] not in {"en", "ru"}:
  164. raise ValueError(f"Unsupported language in context-dependent.tsv: {row['language']}")
  165. renderings = split_aliases(row["possible_renderings"])
  166. aliases = split_aliases(row["aliases"]) + renderings
  167. condition = row["decision_rule"]
  168. if row["context_cues"]:
  169. condition += f" Контекстные признаки: {row['context_cues']}."
  170. records.append(
  171. TermRecord(
  172. english=row["term"] if row["language"] == "en" else "",
  173. russian=row["possible_renderings"],
  174. condition=condition,
  175. aliases=aliases,
  176. source="context guide",
  177. priority=190,
  178. )
  179. )
  180. return records
  181. def load_custom_glossary(path: Path, source: str, priority: int) -> list[TermRecord]:
  182. rows = read_tsv(path, {"english", "russian", "sense", "aliases"})
  183. records: list[TermRecord] = []
  184. seen: set[str] = set()
  185. for row in rows:
  186. if not row["english"] or not row["russian"] or not row["sense"]:
  187. raise ValueError(f"Custom glossary has a blank required value: {path}")
  188. key = normalized_text(row["english"])
  189. if key in seen:
  190. raise ValueError(f"Duplicate custom term in {path}: {row['english']}")
  191. seen.add(key)
  192. records.append(
  193. TermRecord(
  194. english=row["english"],
  195. russian=row["russian"],
  196. condition=row["sense"],
  197. aliases=split_aliases(row["aliases"]),
  198. source=source,
  199. priority=priority,
  200. )
  201. )
  202. return records
  203. def load_raw_records(
  204. entries: list[dict[str, object]],
  205. blocked_english: set[str],
  206. blocked_russian: set[str],
  207. ) -> list[TermRecord]:
  208. records: list[TermRecord] = []
  209. for entry in entries:
  210. english = str(entry["en_raw"])
  211. russian = str(entry["ru_raw"])
  212. if normalized_text(english) in blocked_english or normalized_text(russian) in blocked_russian:
  213. continue
  214. records.append(
  215. TermRecord(
  216. english=english,
  217. russian=russian,
  218. condition=(
  219. "Сверить значение, поддомен и носитель свойства с исходным текстом; "
  220. "пара Scientia сама по себе не утверждает выбор."
  221. ),
  222. aliases=(),
  223. source="Scientia",
  224. priority=100,
  225. raw_candidate=True,
  226. )
  227. )
  228. return records
  229. def record_forms(record: TermRecord, include_raw_single: bool) -> set[tuple[str, ...]]:
  230. values: list[str] = []
  231. for value in (record.english, record.russian, *record.aliases):
  232. values.extend(lexical_variants(value))
  233. forms = {normalized_tokens(value) for value in values if normalized_tokens(value)}
  234. if record.raw_candidate and not include_raw_single:
  235. forms = {form for form in forms if len(form) >= 2}
  236. return forms
  237. def find_spans(tokens: tuple[str, ...], needle: tuple[str, ...]) -> Iterable[tuple[int, int]]:
  238. size = len(needle)
  239. if not size or size > len(tokens):
  240. return
  241. for start in range(len(tokens) - size + 1):
  242. if tokens[start : start + size] == needle:
  243. yield start, start + size
  244. def choose_non_overlapping(hits: list[Hit]) -> list[Hit]:
  245. # Longest phrase wins; project/company authority breaks equal-span ties.
  246. ranked = sorted(
  247. hits,
  248. key=lambda hit: (
  249. -(hit.end - hit.start),
  250. -hit.record.priority,
  251. hit.start,
  252. normalized_text(hit.record.english or hit.record.russian),
  253. ),
  254. )
  255. selected: list[Hit] = []
  256. occupied: set[int] = set()
  257. for hit in ranked:
  258. positions = set(range(hit.start, hit.end))
  259. if positions & occupied:
  260. continue
  261. selected.append(hit)
  262. occupied.update(positions)
  263. return sorted(selected, key=lambda hit: (hit.start, hit.end))
  264. def scan_text(
  265. text: str,
  266. records: list[TermRecord],
  267. include_raw_single: bool,
  268. explicit_label: str | None = None,
  269. pending_forms: set[tuple[str, ...]] | None = None,
  270. ) -> list[Hit]:
  271. tokens = normalized_tokens(text)
  272. pending_spans = [
  273. span
  274. for form in (pending_forms or set())
  275. for span in find_spans(tokens, form)
  276. ]
  277. hits: list[Hit] = []
  278. for record in records:
  279. for form in record_forms(record, include_raw_single):
  280. for start, end in find_spans(tokens, form):
  281. hit_length = end - start
  282. if any(
  283. pending_end - pending_start > hit_length
  284. and start < pending_end
  285. and pending_start < end
  286. for pending_start, pending_end in pending_spans
  287. ):
  288. continue
  289. matched = explicit_label or " ".join(tokens[start:end])
  290. hits.append(Hit(start, end, record, matched))
  291. return choose_non_overlapping(hits)
  292. def fuzzy_raw_matches(query: str, raw_records: list[TermRecord]) -> list[TermRecord]:
  293. query_tokens = normalized_tokens(query)
  294. if not query_tokens:
  295. return []
  296. ranked: list[tuple[int, int, TermRecord]] = []
  297. query_set = set(query_tokens)
  298. for record in raw_records:
  299. best = 0
  300. best_length = 0
  301. for form in record_forms(record, include_raw_single=True):
  302. if form == query_tokens:
  303. score = 100
  304. elif len(query_tokens) <= len(form) and any(
  305. form[index : index + len(query_tokens)] == query_tokens
  306. for index in range(len(form) - len(query_tokens) + 1)
  307. ):
  308. score = 90
  309. elif len(query_tokens) >= 2 and query_set.issubset(set(form)):
  310. score = 70
  311. else:
  312. score = 0
  313. if score > best:
  314. best = score
  315. best_length = len(form)
  316. if best:
  317. ranked.append((best, best_length, record))
  318. ranked.sort(
  319. key=lambda item: (
  320. -item[0],
  321. item[1],
  322. normalized_text(item[2].english),
  323. )
  324. )
  325. return [item[2] for item in ranked[:3]]
  326. def escape_markdown(value: str) -> str:
  327. return value.replace("|", "\\|").replace("\r", " ").replace("\n", " ").strip()
  328. def serialize_hits(hits: list[Hit], diagnostic: bool = False) -> list[dict[str, str]]:
  329. rows: list[dict[str, str]] = []
  330. for hit in hits:
  331. row = {
  332. "matched": hit.matched,
  333. "english": hit.record.english,
  334. "russian": hit.record.russian,
  335. "condition": hit.record.condition,
  336. }
  337. if diagnostic:
  338. row["source"] = hit.record.source
  339. rows.append(row)
  340. return rows
  341. def print_markdown(rows: list[dict[str, str]]) -> None:
  342. if not rows:
  343. print("Подходящих терминологических записей не найдено.")
  344. return
  345. print("| Найдено | English | Русский / варианты | Условие выбора |")
  346. print("|---|---|---|---|")
  347. for row in rows:
  348. print(
  349. "| "
  350. + " | ".join(
  351. escape_markdown(row[key])
  352. for key in ("matched", "english", "russian", "condition")
  353. )
  354. + " |"
  355. )
  356. def build_parser() -> argparse.ArgumentParser:
  357. parser = argparse.ArgumentParser(
  358. description=(
  359. "Select relevant approved, context-dependent and Scientia terminology without "
  360. "loading the full snapshot into model context."
  361. )
  362. )
  363. parser.add_argument("--query", action="append", default=[], help="A Russian or English term")
  364. parser.add_argument("--text", action="append", default=[], help="Source text to scan")
  365. parser.add_argument(
  366. "--text-file",
  367. action="append",
  368. default=[],
  369. type=Path,
  370. help="UTF-8 source file to scan",
  371. )
  372. parser.add_argument(
  373. "--company-glossary",
  374. action="append",
  375. default=[],
  376. type=Path,
  377. help="Approved company TSV using the project template schema",
  378. )
  379. parser.add_argument(
  380. "--project-glossary",
  381. action="append",
  382. default=[],
  383. type=Path,
  384. help="Approved project TSV using runtime/project/template.tsv",
  385. )
  386. parser.add_argument("--limit", type=int, default=12, help="Global result limit (1-50)")
  387. parser.add_argument("--format", choices=["markdown", "json"], default="markdown")
  388. parser.add_argument(
  389. "--diagnostic",
  390. action="store_true",
  391. help="Include glossary origin; allowed only with --format json",
  392. )
  393. parser.add_argument(
  394. "--include-raw-candidates",
  395. action="store_true",
  396. help=(
  397. "Research mode: include unapproved raw Scientia candidate pairs. "
  398. "By default only approved and context-dependent records are returned."
  399. ),
  400. )
  401. parser.add_argument(
  402. "--no-raw-candidates",
  403. action="store_true",
  404. help=argparse.SUPPRESS,
  405. )
  406. return parser
  407. def main() -> int:
  408. args = build_parser().parse_args()
  409. if not (args.query or args.text or args.text_file):
  410. raise ValueError("Provide at least one --query, --text or --text-file")
  411. if len(args.query) > 30:
  412. raise ValueError("At most 30 explicit queries are allowed per call")
  413. if not 1 <= args.limit <= 50:
  414. raise ValueError("--limit must be between 1 and 50")
  415. if args.diagnostic and args.format != "json":
  416. raise ValueError("--diagnostic is allowed only with --format json")
  417. if any(not query.strip() for query in args.query):
  418. raise ValueError("Queries must not be blank")
  419. if args.include_raw_candidates and args.no_raw_candidates:
  420. raise ValueError(
  421. "--include-raw-candidates conflicts with deprecated --no-raw-candidates"
  422. )
  423. source_texts = [text for text in args.text]
  424. source_texts.extend(path.read_text(encoding="utf-8-sig") for path in args.text_file)
  425. if sum(len(text) for text in (*args.query, *source_texts)) > MAX_INPUT_CHARS:
  426. raise ValueError(f"Combined source text exceeds {MAX_INPUT_CHARS} characters")
  427. entries = load_snapshot()
  428. blocked_english, blocked_russian, pending_forms = load_pending_keys(entries)
  429. raw_records = load_raw_records(entries, blocked_english, blocked_russian)
  430. records: list[TermRecord] = []
  431. for path in args.project_glossary:
  432. records.extend(load_custom_glossary(path, "project glossary", 400))
  433. for path in args.company_glossary:
  434. records.extend(load_custom_glossary(path, "company glossary", 300))
  435. records.extend(load_approved_core(entries))
  436. records.extend(load_context_records())
  437. if args.include_raw_candidates:
  438. records.extend(raw_records)
  439. all_hits: list[Hit] = []
  440. for query in args.query:
  441. query_hits = scan_text(
  442. query,
  443. records,
  444. include_raw_single=len(normalized_tokens(query)) == 1,
  445. explicit_label=query,
  446. pending_forms=pending_forms,
  447. )
  448. if not query_hits and args.include_raw_candidates:
  449. query_has_pending_span = any(
  450. next(find_spans(normalized_tokens(query), form), None) is not None
  451. for form in pending_forms
  452. )
  453. if not query_has_pending_span:
  454. query_hits = [
  455. Hit(0, max(1, len(normalized_tokens(query))), record, query)
  456. for record in fuzzy_raw_matches(query, raw_records)
  457. ]
  458. all_hits.extend(query_hits)
  459. for text in source_texts:
  460. all_hits.extend(
  461. scan_text(
  462. text,
  463. records,
  464. include_raw_single=False,
  465. pending_forms=pending_forms,
  466. )
  467. )
  468. # Keep the first occurrence and the highest-priority decision for each concept.
  469. deduplicated: list[Hit] = []
  470. positions: dict[str, int] = {}
  471. for hit in all_hits:
  472. key = normalized_text(hit.record.english or hit.record.russian)
  473. if key not in positions:
  474. positions[key] = len(deduplicated)
  475. deduplicated.append(hit)
  476. continue
  477. index = positions[key]
  478. if hit.record.priority > deduplicated[index].record.priority:
  479. deduplicated[index] = hit
  480. deduplicated = deduplicated[: args.limit]
  481. rows = serialize_hits(deduplicated, diagnostic=args.diagnostic)
  482. if args.format == "json":
  483. print(json.dumps(rows, ensure_ascii=False, indent=2))
  484. else:
  485. print_markdown(rows)
  486. return 0
  487. if __name__ == "__main__":
  488. raise SystemExit(main())