112 lines
3.4 KiB
Python
112 lines
3.4 KiB
Python
"""llm.source — comment + corpus Doc iterators."""
|
|
|
|
import pytest
|
|
|
|
from bib.item import Item
|
|
from bib.store import Store
|
|
from llm.source import (
|
|
_default_root,
|
|
comment_key_map,
|
|
iter_comment_docs,
|
|
iter_corpus_docs,
|
|
)
|
|
|
|
DOCKET = "CMS-2019-0111"
|
|
CID = f"{DOCKET}-0042"
|
|
|
|
COMBINED = f"""---
|
|
comment_id: {CID}
|
|
docket_id: {DOCKET}
|
|
---
|
|
|
|
We object to the E/M consolidation.
|
|
"""
|
|
|
|
|
|
@pytest.fixture
|
|
def store(tmp_path):
|
|
s = Store(":memory:", storage_dir=tmp_path / "storage")
|
|
key = s.create(
|
|
Item(
|
|
item_type="report",
|
|
title="A comment",
|
|
url=f"https://www.regulations.gov/comment/{CID}",
|
|
abstract="Inline abstract body.",
|
|
)
|
|
)
|
|
for tag in (f"docket:{DOCKET}", "doctype:comment", "year:2019"):
|
|
s.add_tag(key, tag)
|
|
s._comment_key = key # test convenience
|
|
return s
|
|
|
|
|
|
@pytest.fixture
|
|
def root(tmp_path):
|
|
d = tmp_path / DOCKET / CID
|
|
d.mkdir(parents=True)
|
|
(d / "combined.md").write_text(COMBINED)
|
|
return tmp_path
|
|
|
|
|
|
class TestCommentDocs:
|
|
def test_extracted_comment_uses_combined_body(self, store, root):
|
|
docs = list(iter_comment_docs(store, docket=DOCKET, root=root))
|
|
assert len(docs) == 1
|
|
assert docs[0].key == store._comment_key
|
|
assert "E/M consolidation" in docs[0].text
|
|
assert docs[0].metadata == {
|
|
"docket": DOCKET,
|
|
"comment_id": CID,
|
|
"doctype": "comment",
|
|
"year": "2019",
|
|
}
|
|
|
|
def test_unextracted_comment_falls_back_to_abstract(self, store, tmp_path):
|
|
docs = list(iter_comment_docs(store, docket=DOCKET, root=tmp_path))
|
|
assert docs[0].text == "Inline abstract body."
|
|
|
|
def test_docket_filter_excludes_others(self, store, root):
|
|
assert list(iter_comment_docs(store, docket="CMS-2021-0119", root=root)) == []
|
|
|
|
|
|
class TestCommentKeyMap:
|
|
def test_maps_comment_id_to_key_and_year(self, store):
|
|
mapping = comment_key_map(store, DOCKET)
|
|
assert mapping[CID] == (store._comment_key, "2019")
|
|
|
|
|
|
class TestCorpusDocs:
|
|
def test_non_comment_item_with_abstract(self, store):
|
|
key = store.create(
|
|
Item(item_type="rule", title="Final rule", abstract="Rule text.")
|
|
)
|
|
store.add_tag(key, "year:2020")
|
|
docs = list(iter_corpus_docs(store))
|
|
assert [d.key for d in docs] == [key]
|
|
assert docs[0].text == "Rule text."
|
|
assert docs[0].metadata["doctype"] == "rule"
|
|
|
|
def test_comments_excluded_from_corpus(self, store):
|
|
assert all(d.metadata["doctype"] != "comment" for d in iter_corpus_docs(store))
|
|
|
|
def test_attachment_text_included(self, store, tmp_path):
|
|
key = store.create(Item(item_type="rule", title="Rule with attachment"))
|
|
store.add_tag(key, "year:2021")
|
|
att = tmp_path / "letter.txt"
|
|
att.write_text("Attachment body text " * 10) # > 50 chars => status "ok"
|
|
store.attach_file(key, att)
|
|
docs = {d.key: d for d in iter_corpus_docs(store)}
|
|
assert "Attachment body text" in docs[key].text
|
|
|
|
def test_empty_text_item_skipped(self, store):
|
|
key = store.create(Item(item_type="rule", title="No content", abstract=" "))
|
|
store.add_tag(key, "year:2022")
|
|
assert key not in {d.key for d in iter_corpus_docs(store)}
|
|
|
|
|
|
class TestDefaultRoot:
|
|
def test_default_root_under_state_comments(self):
|
|
from conf import ROOT
|
|
|
|
assert _default_root() == ROOT / ".state" / "comments"
|