fix(llm): per-code priority cap only under budget pressure (refs #691 #692)

Ruling B15 (coordinator correction on top of 37a970b): B11's "4
priority rows per code" cap was engaging as a default trim rather than
a budget safety net, which regressed the g2064-g2065 golden entry —
G2064 legitimately has 6 priority-kind lineage rows (replaced_by x3,
created, disappeared) and the cap silently dropped 2 of them even
though the question's whole lineage selection had plenty of room under
the hard cap.

llm.lineage._select_for_prompt now computes the unrestricted selection
(every priority row kept, exactly as before B11) first; only when that
would exceed the hard cap (2 * max_rows) does the per-code cap engage
as a fallback. Under budget, every priority row for every code
survives regardless of count.

Tests: replaced the single "always caps at 4" case with two — 6
priority rows for one code all survive under a generous max_rows (hard
cap 50), the same 6 rows trim to 4 under a tight one (hard cap 4).

Re-ran g2064-g2065-to-99424-99426-rate and ccm-history in-process
against the live replica: both PASS (g2064's missing anchor,
JE7KYBW3 p1111, is back).
This commit is contained in:
kert
2026-09-10 13:53:51 -04:00
parent 37a970ba02
commit ac9cb3ec9e
2 changed files with 77 additions and 22 deletions

View File

@@ -189,32 +189,51 @@ def _guidance_line(g: GuidanceRef) -> str:
def _select_for_prompt( def _select_for_prompt(
events: Sequence[LineageEvent], max_rows: int events: Sequence[LineageEvent], max_rows: int
) -> list[LineageEvent]: ) -> list[LineageEvent]:
"""Every priority-kind event (capped at ``_PRIORITY_PER_CODE_CAP`` """Every priority-kind event, plus the earliest-sorted non-priority
per code — Ruling B11), plus the earliest-sorted non-priority events, up to *max_rows* total — priority rows are never dropped by
events, up to *max_rows* total — priority rows are never dropped the non-priority budget. The whole selection is hard-capped at
by the non-priority budget, but the whole selection is still hard- ``2 * max_rows`` (Ruling B11: a question with many eventful codes
capped at ``2 * max_rows`` (Ruling B11: a question with many could otherwise make priority rows alone blow the prompt).
eventful codes could otherwise make priority rows alone blow the
prompt). Selection is by identity (``id()``), not value equality, Ruling B15: the per-code priority cap (``_PRIORITY_PER_CODE_CAP``,
so it stays correct before FR urls are resolved (several collapsed introduced by B11) is a budget SAFETY NET, not a default trim — it
rows can otherwise compare equal on every other field), and render only engages when keeping every priority row would exceed the hard
order follows *events*' own (chronological) order. Shared by cap. When the unrestricted selection already fits within
``2 * max_rows``, every priority row survives exactly as it did
before B11 (a code with 6 legitimate replaces/replaced_by rows, say,
keeps all 6 as long as the question as a whole has room).
Selection is by identity (``id()``), not value equality, so it
stays correct before FR urls are resolved (several collapsed rows
can otherwise compare equal on every other field), and render order
follows *events*' own (chronological) order. Shared by
``LineageEvidence.prompt_block`` (final render) and ``LineageEvidence.prompt_block`` (final render) and
``lineage_evidence`` (deciding which rows are worth an FR url ``lineage_evidence`` (deciding which rows are worth an FR url
lookup — ruling: resolve links only for what the model will see).""" lookup — ruling: resolve links only for what the model will see)."""
hard_cap = 2 * max_rows
priority_all = [e for e in events if e.kind in _PRIORITY_KINDS]
rest = [e for e in events if e.kind not in _PRIORITY_KINDS]
budget_all = max(max_rows - len(priority_all), 0)
total_all = len(priority_all) + min(len(rest), budget_all)
if total_all <= hard_cap:
# Under budget pressure (Ruling B15) — keep every priority row.
keep = {id(e) for e in priority_all} | {id(e) for e in rest[:budget_all]}
else:
# Over the hard cap even before any per-code trimming — fall
# back to B11's per-code safety net.
per_code: dict[str, int] = {} per_code: dict[str, int] = {}
priority_ids: set[int] = set() priority_ids: set[int] = set()
for e in events: for e in priority_all:
if e.kind in _PRIORITY_KINDS:
n = per_code.get(e.code, 0) n = per_code.get(e.code, 0)
if n < _PRIORITY_PER_CODE_CAP: if n < _PRIORITY_PER_CODE_CAP:
priority_ids.add(id(e)) priority_ids.add(id(e))
per_code[e.code] = n + 1 per_code[e.code] = n + 1
rest = [e for e in events if e.kind not in _PRIORITY_KINDS]
budget = max(max_rows - len(priority_ids), 0) budget = max(max_rows - len(priority_ids), 0)
keep = priority_ids | {id(e) for e in rest[:budget]} keep = priority_ids | {id(e) for e in rest[:budget]}
selected = [e for e in events if id(e) in keep] selected = [e for e in events if id(e) in keep]
return selected[: 2 * max_rows] return selected[:hard_cap]
@dataclass(frozen=True) @dataclass(frozen=True)

View File

@@ -727,9 +727,12 @@ class TestPromptBlock:
for kind in ("created", "replaces", "replaced_by", "deleted"): for kind in ("created", "replaces", "replaced_by", "deleted"):
assert kind in rendered assert kind in rendered
def test_priority_rows_capped_at_four_per_code(self): def test_under_budget_every_priority_row_for_a_code_survives(self):
"""Ruling B11: a single code with 6 priority-kind rows only ever """Ruling B15: the per-code priority cap is a budget safety net,
gets 4 of them into the prompt.""" not a default trim — under the hard cap (2 * max_prompt_rows),
a single code's 6 legitimate priority-kind rows (e.g. several
genuine replaces/replaced_by events over the years) all survive,
exactly as before B11 introduced the per-code cap."""
priorities = [ priorities = [
LineageEvent( LineageEvent(
"99490", "99490",
@@ -754,7 +757,40 @@ class TestPromptBlock:
events=tuple(priorities), events=tuple(priorities),
element_diffs=(), element_diffs=(),
guidance=(), guidance=(),
max_prompt_rows=25, max_prompt_rows=25, # hard cap 50 — 6 priority rows fit easily
)
rendered = ev.prompt_block()
assert sum(rendered.count(f"[L{i}]") for i in range(6)) == 6
def test_over_budget_priority_rows_for_a_code_trim_to_four(self):
"""Ruling B15: the same 6 priority-kind rows for one code, but
with a hard cap (2 * max_prompt_rows) too small to hold them —
the per-code cap of 4 engages as the safety net."""
priorities = [
LineageEvent(
"99490",
2020 + i,
"replaces",
(),
(),
f"[L{i}]",
"K",
i,
0,
"",
"fr",
True,
"",
)
for i in range(6)
]
ev = LineageEvidence(
codes=("99490",),
families=(),
events=tuple(priorities),
element_diffs=(),
guidance=(),
max_prompt_rows=2, # hard cap 4 — 6 priority rows don't fit
) )
rendered = ev.prompt_block() rendered = ev.prompt_block()
assert sum(rendered.count(f"[L{i}]") for i in range(6)) == 4 assert sum(rendered.count(f"[L{i}]") for i in range(6)) == 4