Files
stack/tests/bls/test_oews.py
kert 2d5d12b544
All checks were successful
CI / lint (push) Successful in 53s
CI / notebooks-smoke (push) Successful in 1m38s
Deploy / notebooks (push) Has been skipped
CI / test (push) Successful in 2m37s
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 14s
Infra CI / notebooks (push) Successful in 53s
Infra CI / docs (push) Successful in 1m31s
Infra CI / api (push) Successful in 1m0s
Infra CI / mc (push) Successful in 17s
Deploy / report (push) Successful in 15s
Infra CI / llm (push) Successful in 51s
fix(bls): upper-case the OEWS sheet headers — the May 2019 release ships them lower-case, which loaded 0 occupations (refs #695)
2026-09-22 16:29:27 -04:00

391 lines
9.5 KiB
Python

"""bls.oews — OEWS national occupation wages from the annual zips (#695)."""
from __future__ import annotations
import io
import zipfile
import duckdb
import httpx
import openpyxl
import pytest
from bib.store import Store
from bls.oews import (
YEARS,
OewsRow,
cite,
download,
read_national,
read_occupations,
series_id,
to_rows,
write,
zip_url,
)
OLD_COLS = [
"OCC_CODE",
"OCC_TITLE",
"OCC_GROUP",
"TOT_EMP",
"EMP_PRSE",
"H_MEAN",
"A_MEAN",
"MEAN_PRSE",
"H_PCT10",
"H_PCT25",
"H_MEDIAN",
"H_PCT75",
"H_PCT90",
"A_PCT10",
"A_PCT25",
"A_MEDIAN",
"A_PCT75",
"A_PCT90",
"ANNUAL",
"HOURLY",
]
NEW_COLS = [
"AREA",
"AREA_TITLE",
"AREA_TYPE",
"PRIM_STATE",
"NAICS",
"NAICS_TITLE",
"I_GROUP",
"OWN_CODE",
"OCC_CODE",
"OCC_TITLE",
"O_GROUP",
"TOT_EMP",
"EMP_PRSE",
"JOBS_1000",
"LOC_QUOTIENT",
"PCT_TOTAL",
"PCT_RPT",
"H_MEAN",
"A_MEAN",
"MEAN_PRSE",
"H_PCT10",
"H_PCT25",
"H_MEDIAN",
"H_PCT75",
"H_PCT90",
"A_PCT10",
"A_PCT25",
"A_MEDIAN",
"A_PCT75",
"A_PCT90",
"ANNUAL",
"HOURLY",
]
def _xlsx(cols, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(cols)
for r in rows:
ws.append(r)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
def _zip(year: int, cols, rows) -> bytes:
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w") as z:
z.writestr(
f"oesm{year % 100:02d}nat/field_descriptions.xlsx", _xlsx(["a"], [["b"]])
)
z.writestr(
f"oesm{year % 100:02d}nat/national_M{year}_dl.xlsx", _xlsx(cols, rows)
)
return buf.getvalue()
OLD_ROWS = [
[
"00-0000",
"All Occupations",
"total",
132588810,
0.1,
"22.33",
46440,
0.1,
"8.74",
"10.9",
"16.87",
"27.34",
"41.7",
18180,
22670,
35080,
56860,
86730,
None,
None,
],
[
"31-9092",
"Medical Assistants",
"detailed",
571690,
0.8,
"14.8",
30780,
0.3,
"10.23",
"12.09",
"14.24",
"17.12",
"20.5",
21270,
25150,
29610,
35610,
42650,
None,
None,
],
[
"29-1069",
"Physicians and Surgeons, All Other",
"detailed",
308570,
1.2,
"#",
"*",
0.5,
"#",
"#",
"#",
"#",
"#",
"*",
"*",
"*",
"*",
"*",
None,
None,
],
]
NEW_ROWS = [
[
"99",
"U.S.",
"1",
"US",
"000000",
"Cross-industry",
"cross-industry",
"1235",
"31-9092",
"Medical Assistants",
"detailed",
763040,
0.5,
5.025,
1.0,
None,
None,
20.2,
42000,
0.2,
15.1,
17.3,
19.75,
22.1,
25.5,
31400,
36000,
41070,
46000,
53000,
None,
None,
],
[
"99",
"U.S.",
"1",
"US",
"000000",
"Cross-industry",
"cross-industry",
"1235",
"29-1141",
"Registered Nurses",
"detailed",
3175390,
0.4,
20.9,
1.0,
None,
None,
45.42,
94480,
0.3,
30.1,
35.9,
42.1,
52.4,
66.5,
63720,
75000,
86070,
108900,
132680,
None,
None,
],
]
class TestYearsAndIds:
def test_years_and_zip_urls(self):
assert list(YEARS) == list(range(2013, 2025))
assert zip_url(2023) == "https://www.bls.gov/oes/special-requests/oesm23nat.zip"
assert zip_url(2013).endswith("oesm13nat.zip")
def test_series_id_is_the_national_cross_industry_prefix(self):
# OEUN + area 0000000 + industry 000000 + occupation digits; datatype appended by callers
assert series_id("31-9092") == "OEUN000000000000319092"
assert series_id("29-1141", datatype="04") == "OEUN00000000000029114104"
class TestReadNational:
def test_old_layout(self, tmp_path):
p = tmp_path / "oesm13nat.zip"
p.write_bytes(_zip(2013, OLD_COLS, OLD_ROWS))
recs = read_national(p, 2013)
assert len(recs) == 3
assert recs[1]["OCC_CODE"] == "31-9092" and recs[1]["OCC_GROUP"] == "detailed"
def test_new_layout(self, tmp_path):
p = tmp_path / "oesm23nat.zip"
p.write_bytes(_zip(2023, NEW_COLS, NEW_ROWS))
recs = read_national(p, 2023)
assert len(recs) == 2 and recs[0]["O_GROUP"] == "detailed"
def test_lower_case_headers_are_normalised(self, tmp_path):
# the May 2019 file is the one release with lower-case column names
p = tmp_path / "oesm19nat.zip"
p.write_bytes(_zip(2019, [c.lower() for c in NEW_COLS], NEW_ROWS))
recs = read_national(p, 2019)
assert recs[0]["OCC_CODE"] == "31-9092"
assert len(to_rows(2019, recs, item_key="K")) == 2
def test_missing_member(self, tmp_path):
p = tmp_path / "bad.zip"
with zipfile.ZipFile(p, "w") as z:
z.writestr("oesm23nat/other.xlsx", b"x")
with pytest.raises(FileNotFoundError):
read_national(p, 2023)
class TestToRows:
def test_old_layout_rows_and_suppression(self):
rows = to_rows(2013, [dict(zip(OLD_COLS, r)) for r in OLD_ROWS], item_key="K")
assert [r.occ_code for r in rows] == ["00-0000", "31-9092", "29-1069"]
ma = rows[1]
assert isinstance(ma, OewsRow)
assert (ma.year, ma.occ_group, ma.tot_emp) == (2013, "detailed", 571690)
assert ma.h_mean == pytest.approx(14.8) and ma.a_mean == 30780.0
assert ma.a_median == 29610.0 and ma.h_pct90 == pytest.approx(20.5)
assert ma.series_id == "OEUN000000000000319092" and ma.item_key == "K"
phys = rows[2]
assert (
phys.h_mean is None and phys.a_mean is None and phys.a_pct10 is None
) # "#"/"*" → NULL
assert phys.tot_emp == 308570
def test_new_layout_rows(self):
rows = to_rows(2023, [dict(zip(NEW_COLS, r)) for r in NEW_ROWS], item_key="K")
assert [(r.occ_code, r.occ_group) for r in rows] == [
("31-9092", "detailed"),
("29-1141", "detailed"),
]
assert rows[0].a_mean == 42000.0 and rows[0].h_median == pytest.approx(19.75)
assert rows[1].a_pct90 == 132680.0
def test_non_national_rows_are_skipped(self):
state = dict(zip(NEW_COLS, NEW_ROWS[0]))
state.update(AREA="39", AREA_TITLE="Ohio", AREA_TYPE="2")
assert to_rows(2023, [state], item_key="K") == []
class TestDownload:
def test_sends_browser_headers_and_writes_file(self, tmp_path):
seen = {}
def handler(request):
seen["ua"] = request.headers.get("user-agent", "")
seen["accept"] = request.headers.get("accept", "")
seen["url"] = str(request.url)
return httpx.Response(200, content=_zip(2023, NEW_COLS, NEW_ROWS))
client = httpx.Client(transport=httpx.MockTransport(handler))
dest = tmp_path / "oesm23nat.zip"
out = download(client, 2023, dest)
assert out == dest and zipfile.is_zipfile(dest)
assert "Mozilla" in seen["ua"] and "text/html" in seen["accept"]
assert seen["url"] == zip_url(2023)
def test_403_raises(self, tmp_path):
client = httpx.Client(
transport=httpx.MockTransport(lambda r: httpx.Response(403, text="denied"))
)
with pytest.raises(httpx.HTTPStatusError):
download(client, 2023, tmp_path / "x.zip")
def test_non_zip_body_raises(self, tmp_path):
client = httpx.Client(
transport=httpx.MockTransport(
lambda r: httpx.Response(200, text="<html>bot check</html>")
)
)
with pytest.raises(ValueError, match="not a zip"):
download(client, 2023, tmp_path / "x.zip")
@pytest.fixture
def con():
c = duckdb.connect(":memory:")
yield c
c.close()
class TestWriteRead:
def test_write_replaces_year_and_read_filters_occupations(self, con):
r13 = to_rows(2013, [dict(zip(OLD_COLS, r)) for r in OLD_ROWS], item_key="A")
r23 = to_rows(2023, [dict(zip(NEW_COLS, r)) for r in NEW_ROWS], item_key="B")
assert write(con, 2013, r13) == 3 and write(con, 2023, r23) == 2
assert write(con, 2013, r13) == 3
assert con.execute("SELECT count(*) FROM bls.oews").fetchone()[0] == 5
out = read_occupations(con, ["31-9092"])
assert [(r.year, r.a_mean) for r in out] == [(2013, 30780.0), (2023, 42000.0)]
assert (
read_occupations(con, ["31-9092", "29-1141"], years=[2023])[0].occ_code
== "29-1141"
)
class TestCite:
def test_one_source_per_release(self, tmp_path):
s = Store(":memory:", storage_dir=tmp_path / "st")
k1, k2, k3 = cite(s, 2023), cite(s, 2023), cite(s, 2013)
assert k1 == k2 != k3
item = s.get(k1)
assert "May 2023" in item.title and item.url == zip_url(2023)
assert {
"module:bls",
"table:bls.oews",
"source:bls-website",
"year:2023",
} <= set(item.tags)
s.close()