| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546 |
- #!/usr/bin/env python3
- """Select a compact, semantic terminology context without exposing raw metadata."""
- from __future__ import annotations
- import argparse
- import csv
- import hashlib
- import io
- import json
- import re
- import unicodedata
- from dataclasses import dataclass
- from pathlib import Path
- from typing import Iterable
- SKILL_ROOT = Path(__file__).resolve().parent.parent
- TERMINOLOGY_ROOT = SKILL_ROOT / "references" / "terminology"
- RAW_ROOT = TERMINOLOGY_ROOT / "raw" / "scientia"
- MANIFEST_PATH = RAW_ROOT / "manifest.json"
- APPROVED_CORE_PATH = TERMINOLOGY_ROOT / "runtime" / "approved-core.tsv"
- CONTEXT_PATH = TERMINOLOGY_ROOT / "runtime" / "context-dependent.tsv"
- PENDING_PATH = TERMINOLOGY_ROOT / "maintenance" / "pending-corrections.tsv"
- TOKEN_RE = re.compile(r"[0-9a-zа-я]+", re.IGNORECASE)
- MAX_INPUT_CHARS = 5_000_000
- @dataclass(frozen=True)
- class TermRecord:
- english: str
- russian: str
- condition: str
- aliases: tuple[str, ...]
- source: str
- priority: int
- raw_candidate: bool = False
- @dataclass(frozen=True)
- class Hit:
- start: int
- end: int
- record: TermRecord
- matched: str
- def normalized_tokens(value: str) -> tuple[str, ...]:
- value = unicodedata.normalize("NFKC", value).casefold().replace("ё", "е")
- return tuple(TOKEN_RE.findall(value))
- def normalized_text(value: str) -> str:
- return " ".join(normalized_tokens(value))
- def load_snapshot() -> list[dict[str, object]]:
- manifest = json.loads(MANIFEST_PATH.read_text(encoding="utf-8"))
- snapshot_path = RAW_ROOT / str(manifest["raw_file"])
- payload = snapshot_path.read_bytes()
- if hashlib.sha256(payload).hexdigest() != manifest["snapshot_sha256"]:
- raise ValueError("Scientia snapshot hash does not match manifest")
- entries = [
- json.loads(line)
- for line in payload.decode("utf-8").splitlines()
- if line.strip()
- ]
- if len(entries) != manifest["entry_count"]:
- raise ValueError("Scientia snapshot count does not match manifest")
- return entries
- def read_tsv(path: Path, required: set[str]) -> list[dict[str, str]]:
- lines = [
- line
- for line in path.read_text(encoding="utf-8-sig").splitlines()
- if line.strip() and not line.lstrip().startswith("#")
- ]
- if not lines:
- raise ValueError(f"TSV has no header: {path}")
- reader = csv.DictReader(io.StringIO("\n".join(lines)), delimiter="\t")
- fields = set(reader.fieldnames or [])
- missing = required - fields
- if missing:
- raise ValueError(f"TSV {path} is missing columns: {', '.join(sorted(missing))}")
- return [
- {key: (value or "").strip() for key, value in row.items() if key is not None}
- for row in reader
- ]
- def split_aliases(value: str) -> tuple[str, ...]:
- return tuple(item.strip() for item in value.split("|") if item.strip())
- def lexical_variants(value: str) -> tuple[str, ...]:
- """Return useful written variants without inventing translations."""
- variants = [value]
- without_parentheses = re.sub(r"\([^)]*\)", " ", value).strip()
- if without_parentheses:
- variants.append(without_parentheses)
- for group in re.findall(r"\(([^)]*)\)", value):
- variants.extend(re.split(r"[,;/=]+", group))
- variants.extend(re.split(r"[,;/]+", value))
- result: list[str] = []
- seen: set[tuple[str, ...]] = set()
- for variant in variants:
- variant = variant.strip(" =–—-\t")
- tokens = normalized_tokens(variant)
- if tokens and tokens not in seen:
- result.append(variant)
- seen.add(tokens)
- return tuple(result)
- def load_pending_keys(
- entries: list[dict[str, object]],
- ) -> tuple[set[str], set[str], set[tuple[str, ...]]]:
- rows = read_tsv(
- PENDING_PATH,
- {"english_raw", "russian_raw", "status"},
- )
- by_english = {normalized_text(str(entry["en_raw"])): entry for entry in entries}
- blocked_english: set[str] = set()
- blocked_russian: set[str] = set()
- pending_forms: set[tuple[str, ...]] = set()
- for row in rows:
- if row["status"] != "pending":
- raise ValueError("Only pending records belong in pending-corrections.tsv")
- key = normalized_text(row["english_raw"])
- source = by_english.get(key)
- if source is None or str(source["en_raw"]) != row["english_raw"]:
- raise ValueError(f"Pending record is absent from raw snapshot: {row['english_raw']}")
- if str(source["ru_raw"]) != row["russian_raw"]:
- raise ValueError(f"Pending Russian raw value differs from snapshot: {row['english_raw']}")
- blocked_english.add(key)
- blocked_russian.add(normalized_text(row["russian_raw"]))
- for value in (row["english_raw"], row["russian_raw"]):
- for variant in lexical_variants(value):
- tokens = normalized_tokens(variant)
- if tokens:
- pending_forms.add(tokens)
- return blocked_english, blocked_russian, pending_forms
- def load_approved_core(entries: list[dict[str, object]]) -> list[TermRecord]:
- rows = read_tsv(APPROVED_CORE_PATH, {"english", "russian", "sense", "aliases"})
- raw_pairs = {(str(item["en_raw"]), str(item["ru_raw"])) for item in entries}
- records: list[TermRecord] = []
- seen: set[str] = set()
- for row in rows:
- pair = (row["english"], row["russian"])
- if pair not in raw_pairs:
- raise ValueError(
- "approved-core.tsv must contain an exact positive pair from the snapshot: "
- f"{row['english']} — {row['russian']}"
- )
- key = normalized_text(row["english"])
- if not key or key in seen:
- raise ValueError(f"Duplicate or blank approved term: {row['english']}")
- seen.add(key)
- records.append(
- TermRecord(
- english=row["english"],
- russian=row["russian"],
- condition=row["sense"],
- aliases=split_aliases(row["aliases"]),
- source="core glossary",
- priority=200,
- )
- )
- return records
- def load_context_records() -> list[TermRecord]:
- rows = read_tsv(
- CONTEXT_PATH,
- {
- "term",
- "language",
- "possible_renderings",
- "decision_rule",
- "context_cues",
- "aliases",
- },
- )
- records: list[TermRecord] = []
- for row in rows:
- if row["language"] not in {"en", "ru"}:
- raise ValueError(f"Unsupported language in context-dependent.tsv: {row['language']}")
- renderings = split_aliases(row["possible_renderings"])
- aliases = split_aliases(row["aliases"]) + renderings
- condition = row["decision_rule"]
- if row["context_cues"]:
- condition += f" Контекстные признаки: {row['context_cues']}."
- records.append(
- TermRecord(
- english=row["term"] if row["language"] == "en" else "",
- russian=row["possible_renderings"],
- condition=condition,
- aliases=aliases,
- source="context guide",
- priority=190,
- )
- )
- return records
- def load_custom_glossary(path: Path, source: str, priority: int) -> list[TermRecord]:
- rows = read_tsv(path, {"english", "russian", "sense", "aliases"})
- records: list[TermRecord] = []
- seen: set[str] = set()
- for row in rows:
- if not row["english"] or not row["russian"] or not row["sense"]:
- raise ValueError(f"Custom glossary has a blank required value: {path}")
- key = normalized_text(row["english"])
- if key in seen:
- raise ValueError(f"Duplicate custom term in {path}: {row['english']}")
- seen.add(key)
- records.append(
- TermRecord(
- english=row["english"],
- russian=row["russian"],
- condition=row["sense"],
- aliases=split_aliases(row["aliases"]),
- source=source,
- priority=priority,
- )
- )
- return records
- def load_raw_records(
- entries: list[dict[str, object]],
- blocked_english: set[str],
- blocked_russian: set[str],
- ) -> list[TermRecord]:
- records: list[TermRecord] = []
- for entry in entries:
- english = str(entry["en_raw"])
- russian = str(entry["ru_raw"])
- if normalized_text(english) in blocked_english or normalized_text(russian) in blocked_russian:
- continue
- records.append(
- TermRecord(
- english=english,
- russian=russian,
- condition=(
- "Сверить значение, поддомен и носитель свойства с исходным текстом; "
- "пара Scientia сама по себе не утверждает выбор."
- ),
- aliases=(),
- source="Scientia",
- priority=100,
- raw_candidate=True,
- )
- )
- return records
- def record_forms(record: TermRecord, include_raw_single: bool) -> set[tuple[str, ...]]:
- values: list[str] = []
- for value in (record.english, record.russian, *record.aliases):
- values.extend(lexical_variants(value))
- forms = {normalized_tokens(value) for value in values if normalized_tokens(value)}
- if record.raw_candidate and not include_raw_single:
- forms = {form for form in forms if len(form) >= 2}
- return forms
- def find_spans(tokens: tuple[str, ...], needle: tuple[str, ...]) -> Iterable[tuple[int, int]]:
- size = len(needle)
- if not size or size > len(tokens):
- return
- for start in range(len(tokens) - size + 1):
- if tokens[start : start + size] == needle:
- yield start, start + size
- def choose_non_overlapping(hits: list[Hit]) -> list[Hit]:
- # Longest phrase wins; project/company authority breaks equal-span ties.
- ranked = sorted(
- hits,
- key=lambda hit: (
- -(hit.end - hit.start),
- -hit.record.priority,
- hit.start,
- normalized_text(hit.record.english or hit.record.russian),
- ),
- )
- selected: list[Hit] = []
- occupied: set[int] = set()
- for hit in ranked:
- positions = set(range(hit.start, hit.end))
- if positions & occupied:
- continue
- selected.append(hit)
- occupied.update(positions)
- return sorted(selected, key=lambda hit: (hit.start, hit.end))
- def scan_text(
- text: str,
- records: list[TermRecord],
- include_raw_single: bool,
- explicit_label: str | None = None,
- pending_forms: set[tuple[str, ...]] | None = None,
- ) -> list[Hit]:
- tokens = normalized_tokens(text)
- pending_spans = [
- span
- for form in (pending_forms or set())
- for span in find_spans(tokens, form)
- ]
- hits: list[Hit] = []
- for record in records:
- for form in record_forms(record, include_raw_single):
- for start, end in find_spans(tokens, form):
- hit_length = end - start
- if any(
- pending_end - pending_start > hit_length
- and start < pending_end
- and pending_start < end
- for pending_start, pending_end in pending_spans
- ):
- continue
- matched = explicit_label or " ".join(tokens[start:end])
- hits.append(Hit(start, end, record, matched))
- return choose_non_overlapping(hits)
- def fuzzy_raw_matches(query: str, raw_records: list[TermRecord]) -> list[TermRecord]:
- query_tokens = normalized_tokens(query)
- if not query_tokens:
- return []
- ranked: list[tuple[int, int, TermRecord]] = []
- query_set = set(query_tokens)
- for record in raw_records:
- best = 0
- best_length = 0
- for form in record_forms(record, include_raw_single=True):
- if form == query_tokens:
- score = 100
- elif len(query_tokens) <= len(form) and any(
- form[index : index + len(query_tokens)] == query_tokens
- for index in range(len(form) - len(query_tokens) + 1)
- ):
- score = 90
- elif len(query_tokens) >= 2 and query_set.issubset(set(form)):
- score = 70
- else:
- score = 0
- if score > best:
- best = score
- best_length = len(form)
- if best:
- ranked.append((best, best_length, record))
- ranked.sort(
- key=lambda item: (
- -item[0],
- item[1],
- normalized_text(item[2].english),
- )
- )
- return [item[2] for item in ranked[:3]]
- def escape_markdown(value: str) -> str:
- return value.replace("|", "\\|").replace("\r", " ").replace("\n", " ").strip()
- def serialize_hits(hits: list[Hit], diagnostic: bool = False) -> list[dict[str, str]]:
- rows: list[dict[str, str]] = []
- for hit in hits:
- row = {
- "matched": hit.matched,
- "english": hit.record.english,
- "russian": hit.record.russian,
- "condition": hit.record.condition,
- }
- if diagnostic:
- row["source"] = hit.record.source
- rows.append(row)
- return rows
- def print_markdown(rows: list[dict[str, str]]) -> None:
- if not rows:
- print("Подходящих терминологических записей не найдено.")
- return
- print("| Найдено | English | Русский / варианты | Условие выбора |")
- print("|---|---|---|---|")
- for row in rows:
- print(
- "| "
- + " | ".join(
- escape_markdown(row[key])
- for key in ("matched", "english", "russian", "condition")
- )
- + " |"
- )
- def build_parser() -> argparse.ArgumentParser:
- parser = argparse.ArgumentParser(
- description=(
- "Select relevant approved, context-dependent and Scientia terminology without "
- "loading the full snapshot into model context."
- )
- )
- parser.add_argument("--query", action="append", default=[], help="A Russian or English term")
- parser.add_argument("--text", action="append", default=[], help="Source text to scan")
- parser.add_argument(
- "--text-file",
- action="append",
- default=[],
- type=Path,
- help="UTF-8 source file to scan",
- )
- parser.add_argument(
- "--company-glossary",
- action="append",
- default=[],
- type=Path,
- help="Approved company TSV using the project template schema",
- )
- parser.add_argument(
- "--project-glossary",
- action="append",
- default=[],
- type=Path,
- help="Approved project TSV using runtime/project/template.tsv",
- )
- parser.add_argument("--limit", type=int, default=12, help="Global result limit (1-50)")
- parser.add_argument("--format", choices=["markdown", "json"], default="markdown")
- parser.add_argument(
- "--diagnostic",
- action="store_true",
- help="Include glossary origin; allowed only with --format json",
- )
- parser.add_argument(
- "--include-raw-candidates",
- action="store_true",
- help=(
- "Research mode: include unapproved raw Scientia candidate pairs. "
- "By default only approved and context-dependent records are returned."
- ),
- )
- parser.add_argument(
- "--no-raw-candidates",
- action="store_true",
- help=argparse.SUPPRESS,
- )
- return parser
- def main() -> int:
- args = build_parser().parse_args()
- if not (args.query or args.text or args.text_file):
- raise ValueError("Provide at least one --query, --text or --text-file")
- if len(args.query) > 30:
- raise ValueError("At most 30 explicit queries are allowed per call")
- if not 1 <= args.limit <= 50:
- raise ValueError("--limit must be between 1 and 50")
- if args.diagnostic and args.format != "json":
- raise ValueError("--diagnostic is allowed only with --format json")
- if any(not query.strip() for query in args.query):
- raise ValueError("Queries must not be blank")
- if args.include_raw_candidates and args.no_raw_candidates:
- raise ValueError(
- "--include-raw-candidates conflicts with deprecated --no-raw-candidates"
- )
- source_texts = [text for text in args.text]
- source_texts.extend(path.read_text(encoding="utf-8-sig") for path in args.text_file)
- if sum(len(text) for text in (*args.query, *source_texts)) > MAX_INPUT_CHARS:
- raise ValueError(f"Combined source text exceeds {MAX_INPUT_CHARS} characters")
- entries = load_snapshot()
- blocked_english, blocked_russian, pending_forms = load_pending_keys(entries)
- raw_records = load_raw_records(entries, blocked_english, blocked_russian)
- records: list[TermRecord] = []
- for path in args.project_glossary:
- records.extend(load_custom_glossary(path, "project glossary", 400))
- for path in args.company_glossary:
- records.extend(load_custom_glossary(path, "company glossary", 300))
- records.extend(load_approved_core(entries))
- records.extend(load_context_records())
- if args.include_raw_candidates:
- records.extend(raw_records)
- all_hits: list[Hit] = []
- for query in args.query:
- query_hits = scan_text(
- query,
- records,
- include_raw_single=len(normalized_tokens(query)) == 1,
- explicit_label=query,
- pending_forms=pending_forms,
- )
- if not query_hits and args.include_raw_candidates:
- query_has_pending_span = any(
- next(find_spans(normalized_tokens(query), form), None) is not None
- for form in pending_forms
- )
- if not query_has_pending_span:
- query_hits = [
- Hit(0, max(1, len(normalized_tokens(query))), record, query)
- for record in fuzzy_raw_matches(query, raw_records)
- ]
- all_hits.extend(query_hits)
- for text in source_texts:
- all_hits.extend(
- scan_text(
- text,
- records,
- include_raw_single=False,
- pending_forms=pending_forms,
- )
- )
- # Keep the first occurrence and the highest-priority decision for each concept.
- deduplicated: list[Hit] = []
- positions: dict[str, int] = {}
- for hit in all_hits:
- key = normalized_text(hit.record.english or hit.record.russian)
- if key not in positions:
- positions[key] = len(deduplicated)
- deduplicated.append(hit)
- continue
- index = positions[key]
- if hit.record.priority > deduplicated[index].record.priority:
- deduplicated[index] = hit
- deduplicated = deduplicated[: args.limit]
- rows = serialize_hits(deduplicated, diagnostic=args.diagnostic)
- if args.format == "json":
- print(json.dumps(rows, ensure_ascii=False, indent=2))
- else:
- print_markdown(rows)
- return 0
- if __name__ == "__main__":
- raise SystemExit(main())
|