fix(pfs): fr-pairs n_total is the rule's total Comment: count (refs #690)

reaction.fr_pairs previously reported n_total equal to n_items (the
matching Response: count, always the same number as the family's pair
count) — it now reports the rule item's TOTAL Comment: paragraph count,
family or not, matching the n_items-out-of-n_total shape the docket
rows already use. _COMMENT_RE/_RESPONSE_RE are tightened to no leading
whitespace and no space before the colon, so they agree exactly with
the LIKE 'Comment:%'/'Response:%' SQL prefilter.

Also moves the code-pattern/code-match helpers pfs.guidance and
pfs.reaction both need (code_pattern/codes_in) into pfs.families next
to find_codes/FR_CITE_RE, as public functions both modules import
instead of reaction reaching into guidance's private names.
This commit is contained in:
kert
2026-09-10 00:56:21 -04:00
parent 31d9f27d11
commit 8e7d3637f7
7 changed files with 105 additions and 60 deletions

View File

@@ -803,9 +803,9 @@ def _(alt, code, con, mo, not_built, pl):
"paragraph pairs CMS itself printed in a rule's preamble that " "paragraph pairs CMS itself printed in a rule's preamble that "
"name a family code — the pre-2017 proxy, since regulations.gov " "name a family code — the pre-2017 proxy, since regulations.gov "
"comment text isn't indexed that far back; `n_items` is the " "comment text isn't indexed that far back; `n_items` is the "
"qualifying-pair count and `n_total` the matching `Response:` " "qualifying-pair count naming the family and `n_total` the "
"paragraphs (always equal to `n_items`, since a pair requires " "rule's total `Comment:` paragraph count (family or not) — the "
"both)." "pool `n_items` is drawn out of."
+ ( + (
"" ""
if _has_stance if _has_stance

View File

@@ -122,10 +122,13 @@ class ReactionRow:
(``pfs.code_reaction``, #690) — ``period_kind`` is ``"docket"`` (``pfs.code_reaction``, #690) — ``period_kind`` is ``"docket"``
(``period`` is a regulations.gov docket id, ``year`` the docket's (``period`` is a regulations.gov docket id, ``year`` the docket's
``comment_end_date`` year) or ``"fr-pairs"`` (``period`` is the FR ``comment_end_date`` year) or ``"fr-pairs"`` (``period`` is the FR
rule ``item_key``, ``year`` its ``rule_year_of``). ``n_items``/ rule ``item_key``, ``year`` its ``rule_year_of``). For a docket row,
``n_total`` mean distinct commenters mentioning the family / total ``n_items``/``n_total`` are distinct commenters mentioning the
distinct commenters for a docket row, or qualifying Comment:/ family / total distinct commenters in that docket. For an fr-pairs
Response: paragraph pairs for an fr-pairs row. ``stance_*`` and row, ``n_items`` is the qualifying Comment:/Response: paragraph
pairs naming the family and ``n_total`` is the rule's total
``Comment:`` paragraph count — every one in the rule, family or not
— the pool ``n_items`` is drawn out of. ``stance_*`` and
``sample_json`` (``[{"item_key": …, "stance": …}]``) stay zero/ ``sample_json`` (``[{"item_key": …, "stance": …}]``) stay zero/
``"[]"`` unless a stance sample was classified for that docket.""" ``"[]"`` unless a stance sample was classified for that docket."""

View File

@@ -38,6 +38,25 @@ def find_codes(text: str) -> tuple[str, ...]:
return tuple(sorted({m.group(0).upper() for m in CODE_RE.finditer(text)})) return tuple(sorted({m.group(0).upper() for m in CODE_RE.finditer(text)}))
def code_pattern(codes: Sequence[str]) -> re.Pattern[str]:
"""A word-boundary alternation matching any of *codes* (upper-cased).
Shared by ``pfs.guidance`` and ``pfs.reaction`` — both scan FR
paragraph text for a specific family's codes, not every code
``CODE_RE`` would match."""
alts = "|".join(re.escape(c.upper()) for c in codes)
return re.compile(rf"\b(?:{alts})\b")
def codes_in(text: str, pattern: re.Pattern[str]) -> tuple[str, ...]:
"""Codes matching *pattern* literally present in *text*, with FR
citations ("91 FR 43842") stripped first — a citation's page number
reads as a CPT code and must never fire a false match (mirrors
``find_codes``)."""
stripped = _FR_CITE_RE.sub(" ", text)
return tuple(sorted({m.group(0) for m in pattern.finditer(stripped)}))
@dataclass(frozen=True) @dataclass(frozen=True)
class Family: class Family:
key: str key: str

View File

@@ -35,7 +35,7 @@ from typing import Any, Sequence
from bib.cfrlink import canonical, item_for, parse_cite from bib.cfrlink import canonical, item_for, parse_cite
from pfs.codetables import GuidanceRow, is_missing_table_error from pfs.codetables import GuidanceRow, is_missing_table_error
from pfs.families import FR_CITE_RE from pfs.families import code_pattern, codes_in
# manual display name (lower-cased, as it appears in FR prose) -> CMS # manual display name (lower-cased, as it appears in FR prose) -> CMS
# publication number. Controller ruling (task-6-context.md): only these # publication number. Controller ruling (task-6-context.md): only these
@@ -222,20 +222,6 @@ def _extract(text: str, store: Any) -> list[tuple[str, str, str]]:
return out return out
def _code_pattern(codes: Sequence[str]) -> re.Pattern[str]:
alts = "|".join(re.escape(c.upper()) for c in codes)
return re.compile(rf"\b(?:{alts})\b")
def _codes_in(text: str, pattern: re.Pattern[str]) -> tuple[str, ...]:
"""Family codes literally present in *text*, with FR citations
("91 FR 43842") stripped first — a citation's page number reads as
a CPT code and must never fire a false match (mirrors
``pfs.families.find_codes``)."""
stripped = FR_CITE_RE.sub(" ", text)
return tuple(sorted({m.group(0) for m in pattern.finditer(stripped)}))
def dedupe(rows: Sequence[GuidanceRow]) -> list[GuidanceRow]: def dedupe(rows: Sequence[GuidanceRow]) -> list[GuidanceRow]:
"""First-seen-wins de-duplication on ``(family, code, kind, """First-seen-wins de-duplication on ``(family, code, kind,
locator, item_key_src, p_id_src)`` — see the module docstring.""" locator, item_key_src, p_id_src)`` — see the module docstring."""
@@ -257,7 +243,7 @@ def harvest(store: Any, codes: Sequence[str], *, family: str) -> list[GuidanceRo
if not codes: if not codes:
return [] return []
con = store._con() # noqa: SLF001 — same pattern as pfs.descriptors con = store._con() # noqa: SLF001 — same pattern as pfs.descriptors
pattern = _code_pattern(codes) pattern = code_pattern(codes)
like_clause = " OR ".join("text LIKE ?" for _ in codes) like_clause = " OR ".join("text LIKE ?" for _ in codes)
params = [f"%{c.upper()}%" for c in codes] params = [f"%{c.upper()}%" for c in codes]
rows = con.execute( rows = con.execute(
@@ -266,7 +252,7 @@ def harvest(store: Any, codes: Sequence[str], *, family: str) -> list[GuidanceRo
).fetchall() ).fetchall()
out: list[GuidanceRow] = [] out: list[GuidanceRow] = []
for item_key, p_id, page, text in rows: for item_key, p_id, page, text in rows:
present = _codes_in(text, pattern) present = codes_in(text, pattern)
if not present: if not present:
continue continue
for kind, locator, resolved in _extract(text, store): for kind, locator, resolved in _extract(text, store):

View File

@@ -16,8 +16,10 @@ Two period kinds, one ``ReactionRow`` shape:
the vocabulary counts as ``"unclear"``. the vocabulary counts as ``"unclear"``.
* ``"fr-pairs"`` — one row per FR rule item, back to 2001: how many * ``"fr-pairs"`` — one row per FR rule item, back to 2001: how many
``Comment:``/``Response:`` paragraph pairs in that rule's ``Comment:``/``Response:`` paragraph pairs in that rule's
``fr_anchors`` name a family code — the pre-2017 proxy, since ``fr_anchors`` name a family code (``n_items``), out of the rule's
regulations.gov comment text itself isn't indexed for rules that old. total ``Comment:`` paragraph count — family or not (``n_total``) —
the pre-2017 proxy, since regulations.gov comment text itself isn't
indexed for rules that old.
``series`` builds both halves for one family (or one ``--code`` ``series`` builds both halves for one family (or one ``--code``
pseudo-family, a single-code family keyed by the code itself) and pseudo-family, a single-code family keyed by the code itself) and
@@ -36,7 +38,7 @@ from typing import Any, Callable, Sequence
from sqlalchemy import text from sqlalchemy import text
from pfs.descriptors import rule_year_of from pfs.descriptors import rule_year_of
from pfs.guidance import _code_pattern, _codes_in from pfs.families import code_pattern, codes_in
#: Closed vocabulary the stance classifier answers from — anything else #: Closed vocabulary the stance classifier answers from — anything else
#: (a refusal, a hedge, garbage) counts as "unclear" (Resolutions). #: (a refusal, a hedge, garbage) counts as "unclear" (Resolutions).
@@ -161,15 +163,24 @@ def classify_stances(
#: "Comment:" / "Response:" paragraph openers — a docket-comment excerpt #: "Comment:" / "Response:" paragraph openers — a docket-comment excerpt
#: reproduced verbatim inside an FR rule's own preamble, always at the #: reproduced verbatim inside an FR rule's own preamble, always at the
#: very start of the paragraph. #: very start of the paragraph. Tightened to no leading whitespace and no
_COMMENT_RE = re.compile(r"^\s*Comment\s*:", re.I) #: space before the colon (case-insensitive, as before) so these agree
_RESPONSE_RE = re.compile(r"^\s*Response\s*:", re.I) #: exactly with the ``LIKE 'Comment:%'``/``LIKE 'Response:%'`` SQL
#: prefilter in ``fr_pairs`` — a looser Python regex than the SQL LIKE
#: would count paragraphs the SQL never even fetched.
_COMMENT_RE = re.compile(r"^Comment:", re.I)
_RESPONSE_RE = re.compile(r"^Response:", re.I)
def fr_pairs(store: Any, codes: Sequence[str]) -> list[tuple[str, int, int, int]]: def fr_pairs(store: Any, codes: Sequence[str]) -> list[tuple[str, int, int, int]]:
"""``(item_key, rule_year, n_comment_paras, n_response_paras)`` per """``(item_key, rule_year, n_pairs, n_comment_paras)`` per rule item —
rule item — the pre-2017 proxy for public reaction, since the pre-2017 proxy for public reaction, since regulations.gov comment
regulations.gov comment text isn't indexed that far back. text isn't indexed that far back. ``n_pairs`` is the count of
qualifying ``Comment:``/``Response:`` pairs naming one of *codes*
(the family's share); ``n_comment_paras`` is the item's TOTAL count
of ``Comment:`` paragraphs — every one in the rule, family or not,
paired or not — the pool the family's share is drawn out of (the
same n_items-out-of-n_total shape the docket rows use).
Ruling A12: a ``Comment:`` paragraph pairs with the *next* Ruling A12: a ``Comment:`` paragraph pairs with the *next*
``Response:`` paragraph in the same item that comes before the next ``Response:`` paragraph in the same item that comes before the next
@@ -179,15 +190,22 @@ def fr_pairs(store: Any, codes: Sequence[str]) -> list[tuple[str, int, int, int]
whole span (comment + continuations + response), not just the whole span (comment + continuations + response), not just the
comment/response paragraphs themselves. A ``Comment:`` with no comment/response paragraphs themselves. A ``Comment:`` with no
``Response:`` before the next ``Comment:`` (or the end of the item) ``Response:`` before the next ``Comment:`` (or the end of the item)
is unpaired and not counted. Rules with zero qualifying pairs are is unpaired and doesn't count toward ``n_pairs`` — it still counts
omitted. toward ``n_comment_paras``, which is unconditional. Rules with zero
qualifying pairs are omitted (``n_comment_paras`` alone never keeps a
rule in the output).
One SQL per family, scoped to ``Comment:``/``Response:`` openers One SQL per family, scoped to ``Comment:``/``Response:`` openers —
plus any paragraph naming one of *codes* (the same LIKE-prefilter unconditionally, every item's, which is what makes ``n_comment_paras``
idiom ``pfs.guidance.harvest`` uses) — safe under the pairing rule a true item-wide total even though the query is family-scoped — plus
above: a dropped paragraph is prose that names no code and opens any paragraph naming one of *codes* (the same LIKE-prefilter idiom
neither a comment nor a response, so it can neither create nor hide ``pfs.guidance.harvest`` uses) — safe under the pairing rule above: a
a pair. dropped paragraph is prose that names no code and opens neither a
comment nor a response, so it can neither create nor hide a pair.
``_COMMENT_RE``/``_RESPONSE_RE`` are anchored exactly like the SQL's
``LIKE 'Comment:%'``/``LIKE 'Response:%'`` (no leading whitespace, no
space before the colon) so the Python pass never counts a paragraph
the SQL wouldn't have fetched as an opener in the first place.
""" """
if not codes: if not codes:
return [] return []
@@ -202,7 +220,7 @@ def fr_pairs(store: Any, codes: Sequence[str]) -> list[tuple[str, int, int, int]
params, params,
).fetchall() ).fetchall()
pattern = _code_pattern(codes) pattern = code_pattern(codes)
by_item: dict[str, list[str]] = {} by_item: dict[str, list[str]] = {}
meta: dict[str, tuple[str, str]] = {} meta: dict[str, tuple[str, str]] = {}
for item_key, _p_id, para_text, title, date_published in rows: for item_key, _p_id, para_text, title, date_published in rows:
@@ -211,8 +229,7 @@ def fr_pairs(store: Any, codes: Sequence[str]) -> list[tuple[str, int, int, int]
out: list[tuple[str, int, int, int]] = [] out: list[tuple[str, int, int, int]] = []
for item_key, paras in by_item.items(): for item_key, paras in by_item.items():
n_comment = 0 n_pairs = 0
n_response = 0
span: list[str] | None = None # None: no comment currently open span: list[str] | None = None # None: no comment currently open
for para_text in paras: for para_text in paras:
if _COMMENT_RE.match(para_text): if _COMMENT_RE.match(para_text):
@@ -223,17 +240,22 @@ def fr_pairs(store: Any, codes: Sequence[str]) -> list[tuple[str, int, int, int]
if _RESPONSE_RE.match(para_text): if _RESPONSE_RE.match(para_text):
if span is not None: if span is not None:
span.append(para_text) span.append(para_text)
if _codes_in(" ".join(span), pattern): if codes_in(" ".join(span), pattern):
n_comment += 1 n_pairs += 1
n_response += 1
span = None span = None
continue continue
if span is not None: if span is not None:
span.append(para_text) # continuation paragraph span.append(para_text) # continuation paragraph
if n_comment: if n_pairs:
n_comment_paras = sum(1 for p in paras if _COMMENT_RE.match(p))
title, date_published = meta[item_key] title, date_published = meta[item_key]
out.append( out.append(
(item_key, rule_year_of(title, date_published), n_comment, n_response) (
item_key,
rule_year_of(title, date_published),
n_pairs,
n_comment_paras,
)
) )
return sorted(out, key=lambda r: (r[1], r[0])) return sorted(out, key=lambda r: (r[1], r[0]))
@@ -292,15 +314,15 @@ def series(
) )
) )
for item_key, rule_year, n_comment, n_response in fr_pairs(store, codes): for item_key, rule_year, n_pairs, n_comment_paras in fr_pairs(store, codes):
rows.append( rows.append(
ReactionRow( ReactionRow(
family=family, family=family,
period=item_key, period=item_key,
period_kind="fr-pairs", period_kind="fr-pairs",
year=rule_year, year=rule_year,
n_items=n_comment, n_items=n_pairs,
n_total=n_response, n_total=n_comment_paras,
stance_support=0, stance_support=0,
stance_oppose=0, stance_oppose=0,
stance_modify=0, stance_modify=0,

View File

@@ -443,8 +443,18 @@ class TestHarvestCpt:
class TestBuild: class TestBuild:
def test_build_tolerates_missing_cpt_tables(self, store_with_anchors, con): def test_build_tolerates_missing_cpt_tables(self, store_with_anchors):
# A bare connection (no ensure_tables!) so `cpt_years` actually
# raises a Catalog "does not exist" error and harvest_cpt's
# except-is_missing_table_error branch is exercised — the `con`
# fixture already has an (empty) pfs.cpt_section via
# ensure_tables, which only exercises the "if not years: return
# []" branch instead.
s, *_rest = store_with_anchors s, *_rest = store_with_anchors
rows = build(con, s, ("99490", "99439"), family="CCM") bare = duckdb.connect(":memory:")
try:
rows = build(bare, s, ("99490", "99439"), family="CCM")
assert rows # the FR-sourced rows still come through assert rows # the FR-sourced rows still come through
assert harvest_cpt(con, s, ("99490",), family="CCM") == [] assert harvest_cpt(bare, s, ("99490",), family="CCM") == []
finally:
bare.close()

View File

@@ -183,9 +183,11 @@ def store_with_continuation_pair():
class TestFrPairs: class TestFrPairs:
def test_counts_only_pairs_naming_a_code(self, store_with_pairs): def test_counts_only_pairs_naming_a_code(self, store_with_pairs):
# n_total is the item's TOTAL Comment: paragraph count (family or
# not) — store_with_pairs has three: ¶10, ¶12, ¶14.
s, item_key = store_with_pairs s, item_key = store_with_pairs
out = fr_pairs(s, ("99490",)) out = fr_pairs(s, ("99490",))
assert out == [(item_key, 2015, 1, 1)] assert out == [(item_key, 2015, 1, 3)]
def test_no_codes_returns_empty(self, store_with_pairs): def test_no_codes_returns_empty(self, store_with_pairs):
s, _item_key = store_with_pairs s, _item_key = store_with_pairs
@@ -201,7 +203,7 @@ class TestFrPairs:
# 99490. # 99490.
s, item_key = store_with_pairs s, item_key = store_with_pairs
out = fr_pairs(s, ("99490",)) out = fr_pairs(s, ("99490",))
assert out == [(item_key, 2015, 1, 1)] assert out == [(item_key, 2015, 1, 3)]
def test_code_named_only_in_a_continuation_paragraph_still_pairs( def test_code_named_only_in_a_continuation_paragraph_still_pairs(
self, store_with_continuation_pair self, store_with_continuation_pair
@@ -210,6 +212,7 @@ class TestFrPairs:
# between Comment: and Response:, and the code match runs over # between Comment: and Response:, and the code match runs over
# the whole span — neither the Comment: nor the Response: # the whole span — neither the Comment: nor the Response:
# paragraph names 99490 here, only the continuation between them. # paragraph names 99490 here, only the continuation between them.
# This item has one Comment: paragraph total, so n_total is 1.
s, item_key = store_with_continuation_pair s, item_key = store_with_continuation_pair
out = fr_pairs(s, ("99490",)) out = fr_pairs(s, ("99490",))
assert out == [(item_key, 2016, 1, 1)] assert out == [(item_key, 2016, 1, 1)]
@@ -235,7 +238,9 @@ class TestFrPairs:
out = fr_pairs(s, ("99490",)) out = fr_pairs(s, ("99490",))
finally: finally:
s.close() s.close()
assert out == [(item_key, 2017, 1, 1)] # Two Comment: paragraphs total in this item (¶1 abandoned, ¶2
# paired), so n_total is 2 even though only one pair qualifies.
assert out == [(item_key, 2017, 1, 2)]
# ── series ────────────────────────────────────────────────────────── # ── series ──────────────────────────────────────────────────────────
@@ -273,7 +278,7 @@ class TestSeries:
item_key, item_key,
2015, 2015,
1, 1,
1, 3,
) )
def test_stance_sample_classifies_and_fills_sample_json(self, store): def test_stance_sample_classifies_and_fills_sample_json(self, store):