Files
iolai26-solve/solver/tables.py
ModelHub XC 5b016c1af1 初始化项目,由ModelHub XC社区提供模型
Model: rpant/iolai26-solve
Source: Original Platform
2026-07-28 09:36:12 +08:00

182 lines
7.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Table-completion solver: answer items that are rows of a paradigm table
with one or more cells missing.
Covers the recurring Linguini patterns:
- fill_blanks with (k) markers in any column ("netkayʼ | (1) | push"),
including damaged rows where a marker merged with text ("*ʔikat | (4) | (5) to tie");
- numbered query rows lacking one column the context table has
("12. gsnqo'qon | foolishness" against context "word | [IPA] | gloss");
- multi-language columns (Proto-Chamic | Phan Rang Cham | Tsat | meaning).
Method per item:
1. strip (k) markers; the remaining non-empty cell texts are the knowns;
2. map knowns to context-table columns by character overlap (greedy);
3. assign the row's markers, in order, to the free columns, preferring the
free column matching the marker's position in the row (first/last);
4. build (known-column -> answer-column) pairs from the table and pick the
most learnable source column by leave-one-out template-translator fit;
5. predict; abstention returns None (router falls through to LLM/fallback).
"""
from __future__ import annotations
from collections import Counter
from typing import Callable, List, Optional, Sequence, Tuple
from .analogy import solve_from_pairs as analogy_vote
from .fallback import fallback_answer
from .preprocess import _BLANK_MARK, Pair, Puzzle, QueryItem
from .template import TemplateTranslator
from .verifier import leave_one_out
def _col_overlap(cell: str, column: Sequence[str], n: int = 3) -> float:
s = "".join(cell.lower().split())
material = " ".join(c.lower() for c in column)
if len(s) < n:
return 1.0 if any(cell.strip() == c.strip() for c in column) else 0.0
grams = {s[i : i + n] for i in range(len(s) - n + 1)}
return sum(1 for g in grams if g in material) / max(len(grams), 1)
def _clean_rows(table: List[List[str]]) -> List[List[str]]:
if not table:
return []
width = Counter(len(r) for r in table).most_common(1)[0][0]
return [r for r in table
if len(r) == width and not any(_BLANK_MARK.search(c) for c in r)]
def pick_table(puzzle: Puzzle, min_width: int = 2) -> List[List[str]]:
best: List[List[str]] = []
for t in puzzle.tables:
rows = _clean_rows(t)
if rows and len(rows[0]) >= min_width and len(rows) > len(best):
best = rows
return best
def _column_pairs(table: List[List[str]], src_col: int, ans_col: int) -> List[Pair]:
out = []
for r in table:
s, t = r[src_col].strip(), r[ans_col].strip()
if s and t and s != "-" and t != "-":
out.append(Pair(src=s, tgt=t, sep="table"))
return out
def _predictor_for(table: List[List[str]], src_col: int, ans_col: int
) -> Tuple[float, Callable[[str], Optional[str]]]:
"""LOO-scored predictor mapping src_col text to ans_col text. Chain:
template translation (multi-word rows), char-level analogy voting
(single-word paradigm columns: a:b :: query:x over all column pairs),
then the echo fallback."""
pairs = _column_pairs(table, src_col, ans_col)
if len(pairs) < 2:
return -1.0, lambda q: None
attested = [(p.src, p.tgt) for p in pairs]
def make(pair_list: List[Pair]) -> Callable[[str], Optional[str]]:
tmpl = TemplateTranslator(pair_list, "to_work")
ana = [(p.src, p.tgt) for p in pair_list]
def predict(q: str) -> Optional[str]:
ans = tmpl.translate(q)
if ans:
return ans
votes = analogy_vote(ana, q)
if votes:
return votes[0]
return fallback_answer(q, pair_list, "to_work")
return predict
def fit(held_in):
keep = set(held_in)
return make([p for p in pairs if (p.src, p.tgt) in keep])
v = leave_one_out(fit, attested)
return v.score, make(pairs)
class TableSolver:
"""Per-puzzle: caches the context table and column predictors."""
def __init__(self, puzzle: Puzzle):
self.table = pick_table(puzzle)
self._pred_cache: dict = {}
@property
def usable(self) -> bool:
return len(self.table) >= 2
def _predictor(self, src_col: int, ans_col: int):
key = (src_col, ans_col)
if key not in self._pred_cache:
self._pred_cache[key] = _predictor_for(self.table, src_col, ans_col)
return self._pred_cache[key]
def solve(self, item: QueryItem) -> Optional[Tuple[str, float]]:
"""Returns (answer, confidence) or None. Confidence is the LOO fit of
the chosen column predictor."""
if not self.usable or not item.row:
return None
n_cols = len(self.table[0])
cols = [[r[c] for r in self.table] for c in range(n_cols)]
# knowns: marker-stripped non-empty cells, with their row position
knowns: List[Tuple[int, str]] = []
marker_pos: List[Tuple[int, str]] = [] # (cell index, marker number)
for ci, cell in enumerate(item.row):
for k in _BLANK_MARK.findall(cell):
marker_pos.append((ci, k))
text = _BLANK_MARK.sub("", cell).strip()
if text and text != "-":
knowns.append((ci, text))
if not marker_pos and item.blank_col is None:
return None
if len(item.row) == n_cols and item.blank_col is not None:
# same shape as the context table: mapping is positional and the
# marker's own column is the answer column
known_col = {ki: ci for ki, (ci, _) in enumerate(knowns)}
ans_col = item.blank_col
else:
# width mismatch (damaged/narrow rows): map knowns to table
# columns by char overlap with a positional prior as tie-break
scored = sorted(
((_col_overlap(t, cols[c]) + 0.01 / (1 + abs(ci - c)), ki, c)
for ki, (ci, t) in enumerate(knowns) for c in range(n_cols)),
reverse=True,
)
known_col = {}
used = set()
for s, ki, c in scored:
if ki in known_col or c in used:
continue
known_col[ki] = c
used.add(c)
free = [c for c in range(n_cols) if c not in used]
if not free:
return None
# markers in row order take free columns in order; a single
# marker at the row's edge prefers the matching edge column
my_marker_idx = next(
(mi for mi, (_, k) in enumerate(marker_pos) if k == item.number), 0)
if len(marker_pos) <= 1 and item.row and len(free) > 1:
cell_idx = marker_pos[0][0] if marker_pos else (item.blank_col or 0)
ans_col = free[-1] if cell_idx >= len(item.row) - 1 else free[0]
else:
ans_col = free[min(my_marker_idx, len(free) - 1)]
# pick the most learnable known column as the source
best_score, best_predict, best_ki = -1.0, None, None
for ki, c in known_col.items():
score, predict = self._predictor(c, ans_col)
if score > best_score:
best_score, best_predict, best_ki = score, predict, ki
if best_predict is None or best_ki is None:
return None
ans = best_predict(knowns[best_ki][1])
return (ans, max(best_score, 0.0)) if ans else None