"""Deterministic interpreter for DSL grammars: generate() (work -> task language) and analyze() (task -> work language). The interpreter is intentionally strict and simple: it executes exactly what the grammar says. If a grammar needs cleverness, the proposer must encode it (e.g. list 'birds' as its own lexicon entry instead of relying on affix machinery). The verifier then selects grammars that this interpreter executes into correct outputs — that closed loop is the whole design. Conventions the proposer prompt establishes: - lexicon glosses are work-language words/phrases (may be multiword); - affix `feature` is the work-language cue it realizes: a function word ("the", "will", "not") or a marker name ("plural") — during generation a feature fires when its cue appears in the work sentence next to the stem; - `order` is a list of pos tags giving target-language constituent order; - rewrites are surface regex applied after morph concatenation (word-level). """ from __future__ import annotations import re from typing import Dict, List, Optional, Tuple from .grammar import Affix, Grammar, LexEntry def _words(s: str) -> List[str]: return [w for w in re.findall(r"[^\s]+", s.strip()) if w] def _clean(w: str) -> str: return w.strip(",;.!?()[]\"'«»").casefold() class Interpreter: def __init__(self, grammar: Grammar): self.g = grammar # gloss index: multiword glosses first (longest match wins) self._gloss_entries: List[Tuple[List[str], LexEntry]] = sorted( (( [_clean(w) for w in _words(e.gloss)], e) for e in grammar.lexicon if e.gloss), key=lambda t: -len(t[0]), ) self._morphs: Dict[str, LexEntry] = {e.morph: e for e in grammar.lexicon} self._affix_by_cue: Dict[str, List[Affix]] = {} for a in grammar.affixes: self._affix_by_cue.setdefault(_clean(a.feature), []).append(a) # ------------------------------------------------------------------ # generation: work-language sentence -> task-language string # ------------------------------------------------------------------ def generate(self, work_sentence: str) -> Optional[str]: toks = [_clean(w) for w in _words(work_sentence)] if not toks: return None n = len(toks) used = [False] * n stems: List[Tuple[int, LexEntry]] = [] # (position of first gloss word, entry) # 1. cover with lexicon glosses, longest first for gloss_words, entry in self._gloss_entries: L = len(gloss_words) i = 0 while i + L <= n: if not any(used[i : i + L]) and toks[i : i + L] == gloss_words: for j in range(i, i + L): used[j] = True stems.append((i, entry)) i += L else: i += 1 if not stems: return None stems.sort(key=lambda t: t[0]) # 2. leftover tokens fire affixes on the nearest eligible stem pending: Dict[int, List[Affix]] = {k: [] for k in range(len(stems))} uncovered = [i for i in range(n) if not used[i]] for i in uncovered: cue = toks[i] for a in self._affix_by_cue.get(cue, []): k = self._nearest_stem(stems, i, a) if k is not None: pending[k].append(a) break # 3. order stems by target constituent order if pos info available idx = list(range(len(stems))) if self.g.order and all(e.pos for _, e in stems): rank = {pos: r for r, pos in enumerate(self.g.order)} idx.sort(key=lambda k: (rank.get(stems[k][1].pos, len(rank)), stems[k][0])) # 4. build surface words: affix attachment then rewrites out_words = [] for k in idx: _, entry = stems[k] w = entry.morph for a in pending[k]: w = self._attach(w, a) w = self._apply_rewrites(w) out_words.append(w) surface = " ".join(out_words) return self._apply_rewrites_sentence(surface) def _nearest_stem(self, stems: List[Tuple[int, LexEntry]], cue_pos: int, affix: Affix) -> Optional[int]: best_k, best_d = None, 10 ** 9 for k, (pos, entry) in enumerate(stems): if affix.trigger and affix.trigger not in (entry.pos, entry.gloss): continue d = abs(pos - cue_pos) if d < best_d: best_k, best_d = k, d return best_k def _attach(self, w: str, a: Affix) -> str: if a.position == "prefix": return a.form + w if a.position == "suffix": return w + a.form if a.position == "circumfix": return a.form + w + (a.form2 or a.form) if a.position == "infix": # after the first vowel-less onset (common infix site: after first C) m = re.match(r"^([^aeiouAEIOU]*)(.*)$", w) return (m.group(1) + a.form + m.group(2)) if m else w + a.form return w def _apply_rewrites(self, w: str) -> str: for r in self.g.rewrites: w = r.apply(w) return w def _apply_rewrites_sentence(self, s: str) -> str: # rewrites with explicit spaces/anchors act at sentence level too for r in self.g.rewrites: if " " in r.pattern or r.pattern.startswith("^") or r.pattern.endswith("$"): s = r.apply(s) return s # ------------------------------------------------------------------ # analysis: task-language sentence -> work-language string # ------------------------------------------------------------------ def analyze(self, task_sentence: str) -> Optional[str]: words = [_clean(w) for w in _words(task_sentence)] if not words: return None rendered: List[Tuple[str, str, str]] = [] # (prefix cues, gloss, suffix cues) any_known = False for w in words: pre, gloss, post, known = self._analyze_word(w) any_known = any_known or known rendered.append((pre, gloss, post)) if not any_known: return None out: List[str] = [] for pre, gloss, post in rendered: for c in pre.split(): out.append(c) out.append(gloss) for c in post.split(): out.append(c) return " ".join(x for x in out if x) def _analyze_word(self, w: str) -> Tuple[str, str, str, bool]: """Decompose one surface word into (prefix cues, stem gloss, suffix cues, matched?). Tries direct lexicon hit, then affix stripping (longest affix first), then returns the word untouched.""" if w in self._morphs: return "", self._morphs[w].gloss, "", True affixes = sorted(self.g.affixes, key=lambda a: -len(a.form)) for a in affixes: if a.position == "prefix" and w.startswith(a.form): pre, gloss, post, ok = self._analyze_word(w[len(a.form):]) if ok: return (self._cue(a) + " " + pre).strip(), gloss, post, True if a.position == "suffix" and w.endswith(a.form): pre, gloss, post, ok = self._analyze_word(w[: len(w) - len(a.form)]) if ok: return pre, gloss, (post + " " + self._cue(a)).strip(), True if a.position == "circumfix" and w.startswith(a.form) and w.endswith(a.form2 or a.form): inner = w[len(a.form): len(w) - len(a.form2 or a.form)] pre, gloss, post, ok = self._analyze_word(inner) if ok: return (self._cue(a) + " " + pre).strip(), gloss, post, True # last resort: greedy stem containment (rewrite rules may have altered edges) for morph, entry in sorted(self._morphs.items(), key=lambda kv: -len(kv[0])): if len(morph) >= 3 and morph in w: return "", entry.gloss, "", True return "", w, "", False @staticmethod def _cue(a: Affix) -> str: """How an affix surfaces in the work-language output: function-word cues are emitted verbatim; abstract markers (PL, PST) are dropped — the proposer should prefer word cues for translatable material.""" cue = a.feature.strip() if cue and cue.isalpha() and cue.casefold() == cue: return cue return ""