#!/usr/bin/env python3 """Select a compact, semantic terminology context without exposing raw metadata.""" from __future__ import annotations import argparse import csv import hashlib import io import json import re import unicodedata from dataclasses import dataclass from pathlib import Path from typing import Iterable SKILL_ROOT = Path(__file__).resolve().parent.parent TERMINOLOGY_ROOT = SKILL_ROOT / "references" / "terminology" RAW_ROOT = TERMINOLOGY_ROOT / "raw" / "scientia" MANIFEST_PATH = RAW_ROOT / "manifest.json" APPROVED_CORE_PATH = TERMINOLOGY_ROOT / "runtime" / "approved-core.tsv" CONTEXT_PATH = TERMINOLOGY_ROOT / "runtime" / "context-dependent.tsv" PENDING_PATH = TERMINOLOGY_ROOT / "maintenance" / "pending-corrections.tsv" TOKEN_RE = re.compile(r"[0-9a-zа-я]+", re.IGNORECASE) MAX_INPUT_CHARS = 5_000_000 @dataclass(frozen=True) class TermRecord: english: str russian: str condition: str aliases: tuple[str, ...] source: str priority: int raw_candidate: bool = False @dataclass(frozen=True) class Hit: start: int end: int record: TermRecord matched: str def normalized_tokens(value: str) -> tuple[str, ...]: value = unicodedata.normalize("NFKC", value).casefold().replace("ё", "е") return tuple(TOKEN_RE.findall(value)) def normalized_text(value: str) -> str: return " ".join(normalized_tokens(value)) def load_snapshot() -> list[dict[str, object]]: manifest = json.loads(MANIFEST_PATH.read_text(encoding="utf-8")) snapshot_path = RAW_ROOT / str(manifest["raw_file"]) payload = snapshot_path.read_bytes() if hashlib.sha256(payload).hexdigest() != manifest["snapshot_sha256"]: raise ValueError("Scientia snapshot hash does not match manifest") entries = [ json.loads(line) for line in payload.decode("utf-8").splitlines() if line.strip() ] if len(entries) != manifest["entry_count"]: raise ValueError("Scientia snapshot count does not match manifest") return entries def read_tsv(path: Path, required: set[str]) -> list[dict[str, str]]: lines = [ line for line in path.read_text(encoding="utf-8-sig").splitlines() if line.strip() and not line.lstrip().startswith("#") ] if not lines: raise ValueError(f"TSV has no header: {path}") reader = csv.DictReader(io.StringIO("\n".join(lines)), delimiter="\t") fields = set(reader.fieldnames or []) missing = required - fields if missing: raise ValueError(f"TSV {path} is missing columns: {', '.join(sorted(missing))}") return [ {key: (value or "").strip() for key, value in row.items() if key is not None} for row in reader ] def split_aliases(value: str) -> tuple[str, ...]: return tuple(item.strip() for item in value.split("|") if item.strip()) def lexical_variants(value: str) -> tuple[str, ...]: """Return useful written variants without inventing translations.""" variants = [value] without_parentheses = re.sub(r"\([^)]*\)", " ", value).strip() if without_parentheses: variants.append(without_parentheses) for group in re.findall(r"\(([^)]*)\)", value): variants.extend(re.split(r"[,;/=]+", group)) variants.extend(re.split(r"[,;/]+", value)) result: list[str] = [] seen: set[tuple[str, ...]] = set() for variant in variants: variant = variant.strip(" =–—-\t") tokens = normalized_tokens(variant) if tokens and tokens not in seen: result.append(variant) seen.add(tokens) return tuple(result) def load_pending_keys( entries: list[dict[str, object]], ) -> tuple[set[str], set[str], set[tuple[str, ...]]]: rows = read_tsv( PENDING_PATH, {"english_raw", "russian_raw", "status"}, ) by_english = {normalized_text(str(entry["en_raw"])): entry for entry in entries} blocked_english: set[str] = set() blocked_russian: set[str] = set() pending_forms: set[tuple[str, ...]] = set() for row in rows: if row["status"] != "pending": raise ValueError("Only pending records belong in pending-corrections.tsv") key = normalized_text(row["english_raw"]) source = by_english.get(key) if source is None or str(source["en_raw"]) != row["english_raw"]: raise ValueError(f"Pending record is absent from raw snapshot: {row['english_raw']}") if str(source["ru_raw"]) != row["russian_raw"]: raise ValueError(f"Pending Russian raw value differs from snapshot: {row['english_raw']}") blocked_english.add(key) blocked_russian.add(normalized_text(row["russian_raw"])) for value in (row["english_raw"], row["russian_raw"]): for variant in lexical_variants(value): tokens = normalized_tokens(variant) if tokens: pending_forms.add(tokens) return blocked_english, blocked_russian, pending_forms def load_approved_core(entries: list[dict[str, object]]) -> list[TermRecord]: rows = read_tsv(APPROVED_CORE_PATH, {"english", "russian", "sense", "aliases"}) raw_pairs = {(str(item["en_raw"]), str(item["ru_raw"])) for item in entries} records: list[TermRecord] = [] seen: set[str] = set() for row in rows: pair = (row["english"], row["russian"]) if pair not in raw_pairs: raise ValueError( "approved-core.tsv must contain an exact positive pair from the snapshot: " f"{row['english']} — {row['russian']}" ) key = normalized_text(row["english"]) if not key or key in seen: raise ValueError(f"Duplicate or blank approved term: {row['english']}") seen.add(key) records.append( TermRecord( english=row["english"], russian=row["russian"], condition=row["sense"], aliases=split_aliases(row["aliases"]), source="core glossary", priority=200, ) ) return records def load_context_records() -> list[TermRecord]: rows = read_tsv( CONTEXT_PATH, { "term", "language", "possible_renderings", "decision_rule", "context_cues", "aliases", }, ) records: list[TermRecord] = [] for row in rows: if row["language"] not in {"en", "ru"}: raise ValueError(f"Unsupported language in context-dependent.tsv: {row['language']}") renderings = split_aliases(row["possible_renderings"]) aliases = split_aliases(row["aliases"]) + renderings condition = row["decision_rule"] if row["context_cues"]: condition += f" Контекстные признаки: {row['context_cues']}." records.append( TermRecord( english=row["term"] if row["language"] == "en" else "", russian=row["possible_renderings"], condition=condition, aliases=aliases, source="context guide", priority=190, ) ) return records def load_custom_glossary(path: Path, source: str, priority: int) -> list[TermRecord]: rows = read_tsv(path, {"english", "russian", "sense", "aliases"}) records: list[TermRecord] = [] seen: set[str] = set() for row in rows: if not row["english"] or not row["russian"] or not row["sense"]: raise ValueError(f"Custom glossary has a blank required value: {path}") key = normalized_text(row["english"]) if key in seen: raise ValueError(f"Duplicate custom term in {path}: {row['english']}") seen.add(key) records.append( TermRecord( english=row["english"], russian=row["russian"], condition=row["sense"], aliases=split_aliases(row["aliases"]), source=source, priority=priority, ) ) return records def load_raw_records( entries: list[dict[str, object]], blocked_english: set[str], blocked_russian: set[str], ) -> list[TermRecord]: records: list[TermRecord] = [] for entry in entries: english = str(entry["en_raw"]) russian = str(entry["ru_raw"]) if normalized_text(english) in blocked_english or normalized_text(russian) in blocked_russian: continue records.append( TermRecord( english=english, russian=russian, condition=( "Сверить значение, поддомен и носитель свойства с исходным текстом; " "пара Scientia сама по себе не утверждает выбор." ), aliases=(), source="Scientia", priority=100, raw_candidate=True, ) ) return records def record_forms(record: TermRecord, include_raw_single: bool) -> set[tuple[str, ...]]: values: list[str] = [] for value in (record.english, record.russian, *record.aliases): values.extend(lexical_variants(value)) forms = {normalized_tokens(value) for value in values if normalized_tokens(value)} if record.raw_candidate and not include_raw_single: forms = {form for form in forms if len(form) >= 2} return forms def find_spans(tokens: tuple[str, ...], needle: tuple[str, ...]) -> Iterable[tuple[int, int]]: size = len(needle) if not size or size > len(tokens): return for start in range(len(tokens) - size + 1): if tokens[start : start + size] == needle: yield start, start + size def choose_non_overlapping(hits: list[Hit]) -> list[Hit]: # Longest phrase wins; project/company authority breaks equal-span ties. ranked = sorted( hits, key=lambda hit: ( -(hit.end - hit.start), -hit.record.priority, hit.start, normalized_text(hit.record.english or hit.record.russian), ), ) selected: list[Hit] = [] occupied: set[int] = set() for hit in ranked: positions = set(range(hit.start, hit.end)) if positions & occupied: continue selected.append(hit) occupied.update(positions) return sorted(selected, key=lambda hit: (hit.start, hit.end)) def scan_text( text: str, records: list[TermRecord], include_raw_single: bool, explicit_label: str | None = None, pending_forms: set[tuple[str, ...]] | None = None, ) -> list[Hit]: tokens = normalized_tokens(text) pending_spans = [ span for form in (pending_forms or set()) for span in find_spans(tokens, form) ] hits: list[Hit] = [] for record in records: for form in record_forms(record, include_raw_single): for start, end in find_spans(tokens, form): hit_length = end - start if any( pending_end - pending_start > hit_length and start < pending_end and pending_start < end for pending_start, pending_end in pending_spans ): continue matched = explicit_label or " ".join(tokens[start:end]) hits.append(Hit(start, end, record, matched)) return choose_non_overlapping(hits) def fuzzy_raw_matches(query: str, raw_records: list[TermRecord]) -> list[TermRecord]: query_tokens = normalized_tokens(query) if not query_tokens: return [] ranked: list[tuple[int, int, TermRecord]] = [] query_set = set(query_tokens) for record in raw_records: best = 0 best_length = 0 for form in record_forms(record, include_raw_single=True): if form == query_tokens: score = 100 elif len(query_tokens) <= len(form) and any( form[index : index + len(query_tokens)] == query_tokens for index in range(len(form) - len(query_tokens) + 1) ): score = 90 elif len(query_tokens) >= 2 and query_set.issubset(set(form)): score = 70 else: score = 0 if score > best: best = score best_length = len(form) if best: ranked.append((best, best_length, record)) ranked.sort( key=lambda item: ( -item[0], item[1], normalized_text(item[2].english), ) ) return [item[2] for item in ranked[:3]] def escape_markdown(value: str) -> str: return value.replace("|", "\\|").replace("\r", " ").replace("\n", " ").strip() def serialize_hits(hits: list[Hit], diagnostic: bool = False) -> list[dict[str, str]]: rows: list[dict[str, str]] = [] for hit in hits: row = { "matched": hit.matched, "english": hit.record.english, "russian": hit.record.russian, "condition": hit.record.condition, } if diagnostic: row["source"] = hit.record.source rows.append(row) return rows def print_markdown(rows: list[dict[str, str]]) -> None: if not rows: print("Подходящих терминологических записей не найдено.") return print("| Найдено | English | Русский / варианты | Условие выбора |") print("|---|---|---|---|") for row in rows: print( "| " + " | ".join( escape_markdown(row[key]) for key in ("matched", "english", "russian", "condition") ) + " |" ) def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( description=( "Select relevant approved, context-dependent and Scientia terminology without " "loading the full snapshot into model context." ) ) parser.add_argument("--query", action="append", default=[], help="A Russian or English term") parser.add_argument("--text", action="append", default=[], help="Source text to scan") parser.add_argument( "--text-file", action="append", default=[], type=Path, help="UTF-8 source file to scan", ) parser.add_argument( "--company-glossary", action="append", default=[], type=Path, help="Approved company TSV using the project template schema", ) parser.add_argument( "--project-glossary", action="append", default=[], type=Path, help="Approved project TSV using runtime/project/template.tsv", ) parser.add_argument("--limit", type=int, default=12, help="Global result limit (1-50)") parser.add_argument("--format", choices=["markdown", "json"], default="markdown") parser.add_argument( "--diagnostic", action="store_true", help="Include glossary origin; allowed only with --format json", ) parser.add_argument( "--include-raw-candidates", action="store_true", help=( "Research mode: include unapproved raw Scientia candidate pairs. " "By default only approved and context-dependent records are returned." ), ) parser.add_argument( "--no-raw-candidates", action="store_true", help=argparse.SUPPRESS, ) return parser def main() -> int: args = build_parser().parse_args() if not (args.query or args.text or args.text_file): raise ValueError("Provide at least one --query, --text or --text-file") if len(args.query) > 30: raise ValueError("At most 30 explicit queries are allowed per call") if not 1 <= args.limit <= 50: raise ValueError("--limit must be between 1 and 50") if args.diagnostic and args.format != "json": raise ValueError("--diagnostic is allowed only with --format json") if any(not query.strip() for query in args.query): raise ValueError("Queries must not be blank") if args.include_raw_candidates and args.no_raw_candidates: raise ValueError( "--include-raw-candidates conflicts with deprecated --no-raw-candidates" ) source_texts = [text for text in args.text] source_texts.extend(path.read_text(encoding="utf-8-sig") for path in args.text_file) if sum(len(text) for text in (*args.query, *source_texts)) > MAX_INPUT_CHARS: raise ValueError(f"Combined source text exceeds {MAX_INPUT_CHARS} characters") entries = load_snapshot() blocked_english, blocked_russian, pending_forms = load_pending_keys(entries) raw_records = load_raw_records(entries, blocked_english, blocked_russian) records: list[TermRecord] = [] for path in args.project_glossary: records.extend(load_custom_glossary(path, "project glossary", 400)) for path in args.company_glossary: records.extend(load_custom_glossary(path, "company glossary", 300)) records.extend(load_approved_core(entries)) records.extend(load_context_records()) if args.include_raw_candidates: records.extend(raw_records) all_hits: list[Hit] = [] for query in args.query: query_hits = scan_text( query, records, include_raw_single=len(normalized_tokens(query)) == 1, explicit_label=query, pending_forms=pending_forms, ) if not query_hits and args.include_raw_candidates: query_has_pending_span = any( next(find_spans(normalized_tokens(query), form), None) is not None for form in pending_forms ) if not query_has_pending_span: query_hits = [ Hit(0, max(1, len(normalized_tokens(query))), record, query) for record in fuzzy_raw_matches(query, raw_records) ] all_hits.extend(query_hits) for text in source_texts: all_hits.extend( scan_text( text, records, include_raw_single=False, pending_forms=pending_forms, ) ) # Keep the first occurrence and the highest-priority decision for each concept. deduplicated: list[Hit] = [] positions: dict[str, int] = {} for hit in all_hits: key = normalized_text(hit.record.english or hit.record.russian) if key not in positions: positions[key] = len(deduplicated) deduplicated.append(hit) continue index = positions[key] if hit.record.priority > deduplicated[index].record.priority: deduplicated[index] = hit deduplicated = deduplicated[: args.limit] rows = serialize_hits(deduplicated, diagnostic=args.diagnostic) if args.format == "json": print(json.dumps(rows, ensure_ascii=False, indent=2)) else: print_markdown(rows) return 0 if __name__ == "__main__": raise SystemExit(main())