All checks were successful
CI / lint (push) Successful in 53s
CI / notebooks-smoke (push) Successful in 1m38s
Deploy / notebooks (push) Has been skipped
CI / test (push) Successful in 2m37s
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 14s
Infra CI / notebooks (push) Successful in 53s
Infra CI / docs (push) Successful in 1m31s
Infra CI / api (push) Successful in 1m0s
Infra CI / mc (push) Successful in 17s
Deploy / report (push) Successful in 15s
Infra CI / llm (push) Successful in 51s
391 lines
9.5 KiB
Python
391 lines
9.5 KiB
Python
"""bls.oews — OEWS national occupation wages from the annual zips (#695)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import zipfile
|
|
|
|
import duckdb
|
|
import httpx
|
|
import openpyxl
|
|
import pytest
|
|
|
|
from bib.store import Store
|
|
from bls.oews import (
|
|
YEARS,
|
|
OewsRow,
|
|
cite,
|
|
download,
|
|
read_national,
|
|
read_occupations,
|
|
series_id,
|
|
to_rows,
|
|
write,
|
|
zip_url,
|
|
)
|
|
|
|
OLD_COLS = [
|
|
"OCC_CODE",
|
|
"OCC_TITLE",
|
|
"OCC_GROUP",
|
|
"TOT_EMP",
|
|
"EMP_PRSE",
|
|
"H_MEAN",
|
|
"A_MEAN",
|
|
"MEAN_PRSE",
|
|
"H_PCT10",
|
|
"H_PCT25",
|
|
"H_MEDIAN",
|
|
"H_PCT75",
|
|
"H_PCT90",
|
|
"A_PCT10",
|
|
"A_PCT25",
|
|
"A_MEDIAN",
|
|
"A_PCT75",
|
|
"A_PCT90",
|
|
"ANNUAL",
|
|
"HOURLY",
|
|
]
|
|
NEW_COLS = [
|
|
"AREA",
|
|
"AREA_TITLE",
|
|
"AREA_TYPE",
|
|
"PRIM_STATE",
|
|
"NAICS",
|
|
"NAICS_TITLE",
|
|
"I_GROUP",
|
|
"OWN_CODE",
|
|
"OCC_CODE",
|
|
"OCC_TITLE",
|
|
"O_GROUP",
|
|
"TOT_EMP",
|
|
"EMP_PRSE",
|
|
"JOBS_1000",
|
|
"LOC_QUOTIENT",
|
|
"PCT_TOTAL",
|
|
"PCT_RPT",
|
|
"H_MEAN",
|
|
"A_MEAN",
|
|
"MEAN_PRSE",
|
|
"H_PCT10",
|
|
"H_PCT25",
|
|
"H_MEDIAN",
|
|
"H_PCT75",
|
|
"H_PCT90",
|
|
"A_PCT10",
|
|
"A_PCT25",
|
|
"A_MEDIAN",
|
|
"A_PCT75",
|
|
"A_PCT90",
|
|
"ANNUAL",
|
|
"HOURLY",
|
|
]
|
|
|
|
|
|
def _xlsx(cols, rows) -> bytes:
|
|
wb = openpyxl.Workbook()
|
|
ws = wb.active
|
|
ws.append(cols)
|
|
for r in rows:
|
|
ws.append(r)
|
|
buf = io.BytesIO()
|
|
wb.save(buf)
|
|
return buf.getvalue()
|
|
|
|
|
|
def _zip(year: int, cols, rows) -> bytes:
|
|
buf = io.BytesIO()
|
|
with zipfile.ZipFile(buf, "w") as z:
|
|
z.writestr(
|
|
f"oesm{year % 100:02d}nat/field_descriptions.xlsx", _xlsx(["a"], [["b"]])
|
|
)
|
|
z.writestr(
|
|
f"oesm{year % 100:02d}nat/national_M{year}_dl.xlsx", _xlsx(cols, rows)
|
|
)
|
|
return buf.getvalue()
|
|
|
|
|
|
OLD_ROWS = [
|
|
[
|
|
"00-0000",
|
|
"All Occupations",
|
|
"total",
|
|
132588810,
|
|
0.1,
|
|
"22.33",
|
|
46440,
|
|
0.1,
|
|
"8.74",
|
|
"10.9",
|
|
"16.87",
|
|
"27.34",
|
|
"41.7",
|
|
18180,
|
|
22670,
|
|
35080,
|
|
56860,
|
|
86730,
|
|
None,
|
|
None,
|
|
],
|
|
[
|
|
"31-9092",
|
|
"Medical Assistants",
|
|
"detailed",
|
|
571690,
|
|
0.8,
|
|
"14.8",
|
|
30780,
|
|
0.3,
|
|
"10.23",
|
|
"12.09",
|
|
"14.24",
|
|
"17.12",
|
|
"20.5",
|
|
21270,
|
|
25150,
|
|
29610,
|
|
35610,
|
|
42650,
|
|
None,
|
|
None,
|
|
],
|
|
[
|
|
"29-1069",
|
|
"Physicians and Surgeons, All Other",
|
|
"detailed",
|
|
308570,
|
|
1.2,
|
|
"#",
|
|
"*",
|
|
0.5,
|
|
"#",
|
|
"#",
|
|
"#",
|
|
"#",
|
|
"#",
|
|
"*",
|
|
"*",
|
|
"*",
|
|
"*",
|
|
"*",
|
|
None,
|
|
None,
|
|
],
|
|
]
|
|
NEW_ROWS = [
|
|
[
|
|
"99",
|
|
"U.S.",
|
|
"1",
|
|
"US",
|
|
"000000",
|
|
"Cross-industry",
|
|
"cross-industry",
|
|
"1235",
|
|
"31-9092",
|
|
"Medical Assistants",
|
|
"detailed",
|
|
763040,
|
|
0.5,
|
|
5.025,
|
|
1.0,
|
|
None,
|
|
None,
|
|
20.2,
|
|
42000,
|
|
0.2,
|
|
15.1,
|
|
17.3,
|
|
19.75,
|
|
22.1,
|
|
25.5,
|
|
31400,
|
|
36000,
|
|
41070,
|
|
46000,
|
|
53000,
|
|
None,
|
|
None,
|
|
],
|
|
[
|
|
"99",
|
|
"U.S.",
|
|
"1",
|
|
"US",
|
|
"000000",
|
|
"Cross-industry",
|
|
"cross-industry",
|
|
"1235",
|
|
"29-1141",
|
|
"Registered Nurses",
|
|
"detailed",
|
|
3175390,
|
|
0.4,
|
|
20.9,
|
|
1.0,
|
|
None,
|
|
None,
|
|
45.42,
|
|
94480,
|
|
0.3,
|
|
30.1,
|
|
35.9,
|
|
42.1,
|
|
52.4,
|
|
66.5,
|
|
63720,
|
|
75000,
|
|
86070,
|
|
108900,
|
|
132680,
|
|
None,
|
|
None,
|
|
],
|
|
]
|
|
|
|
|
|
class TestYearsAndIds:
|
|
def test_years_and_zip_urls(self):
|
|
assert list(YEARS) == list(range(2013, 2025))
|
|
assert zip_url(2023) == "https://www.bls.gov/oes/special-requests/oesm23nat.zip"
|
|
assert zip_url(2013).endswith("oesm13nat.zip")
|
|
|
|
def test_series_id_is_the_national_cross_industry_prefix(self):
|
|
# OEUN + area 0000000 + industry 000000 + occupation digits; datatype appended by callers
|
|
assert series_id("31-9092") == "OEUN000000000000319092"
|
|
assert series_id("29-1141", datatype="04") == "OEUN00000000000029114104"
|
|
|
|
|
|
class TestReadNational:
|
|
def test_old_layout(self, tmp_path):
|
|
p = tmp_path / "oesm13nat.zip"
|
|
p.write_bytes(_zip(2013, OLD_COLS, OLD_ROWS))
|
|
recs = read_national(p, 2013)
|
|
assert len(recs) == 3
|
|
assert recs[1]["OCC_CODE"] == "31-9092" and recs[1]["OCC_GROUP"] == "detailed"
|
|
|
|
def test_new_layout(self, tmp_path):
|
|
p = tmp_path / "oesm23nat.zip"
|
|
p.write_bytes(_zip(2023, NEW_COLS, NEW_ROWS))
|
|
recs = read_national(p, 2023)
|
|
assert len(recs) == 2 and recs[0]["O_GROUP"] == "detailed"
|
|
|
|
def test_lower_case_headers_are_normalised(self, tmp_path):
|
|
# the May 2019 file is the one release with lower-case column names
|
|
p = tmp_path / "oesm19nat.zip"
|
|
p.write_bytes(_zip(2019, [c.lower() for c in NEW_COLS], NEW_ROWS))
|
|
recs = read_national(p, 2019)
|
|
assert recs[0]["OCC_CODE"] == "31-9092"
|
|
assert len(to_rows(2019, recs, item_key="K")) == 2
|
|
|
|
def test_missing_member(self, tmp_path):
|
|
p = tmp_path / "bad.zip"
|
|
with zipfile.ZipFile(p, "w") as z:
|
|
z.writestr("oesm23nat/other.xlsx", b"x")
|
|
with pytest.raises(FileNotFoundError):
|
|
read_national(p, 2023)
|
|
|
|
|
|
class TestToRows:
|
|
def test_old_layout_rows_and_suppression(self):
|
|
rows = to_rows(2013, [dict(zip(OLD_COLS, r)) for r in OLD_ROWS], item_key="K")
|
|
assert [r.occ_code for r in rows] == ["00-0000", "31-9092", "29-1069"]
|
|
ma = rows[1]
|
|
assert isinstance(ma, OewsRow)
|
|
assert (ma.year, ma.occ_group, ma.tot_emp) == (2013, "detailed", 571690)
|
|
assert ma.h_mean == pytest.approx(14.8) and ma.a_mean == 30780.0
|
|
assert ma.a_median == 29610.0 and ma.h_pct90 == pytest.approx(20.5)
|
|
assert ma.series_id == "OEUN000000000000319092" and ma.item_key == "K"
|
|
phys = rows[2]
|
|
assert (
|
|
phys.h_mean is None and phys.a_mean is None and phys.a_pct10 is None
|
|
) # "#"/"*" → NULL
|
|
assert phys.tot_emp == 308570
|
|
|
|
def test_new_layout_rows(self):
|
|
rows = to_rows(2023, [dict(zip(NEW_COLS, r)) for r in NEW_ROWS], item_key="K")
|
|
assert [(r.occ_code, r.occ_group) for r in rows] == [
|
|
("31-9092", "detailed"),
|
|
("29-1141", "detailed"),
|
|
]
|
|
assert rows[0].a_mean == 42000.0 and rows[0].h_median == pytest.approx(19.75)
|
|
assert rows[1].a_pct90 == 132680.0
|
|
|
|
def test_non_national_rows_are_skipped(self):
|
|
state = dict(zip(NEW_COLS, NEW_ROWS[0]))
|
|
state.update(AREA="39", AREA_TITLE="Ohio", AREA_TYPE="2")
|
|
assert to_rows(2023, [state], item_key="K") == []
|
|
|
|
|
|
class TestDownload:
|
|
def test_sends_browser_headers_and_writes_file(self, tmp_path):
|
|
seen = {}
|
|
|
|
def handler(request):
|
|
seen["ua"] = request.headers.get("user-agent", "")
|
|
seen["accept"] = request.headers.get("accept", "")
|
|
seen["url"] = str(request.url)
|
|
return httpx.Response(200, content=_zip(2023, NEW_COLS, NEW_ROWS))
|
|
|
|
client = httpx.Client(transport=httpx.MockTransport(handler))
|
|
dest = tmp_path / "oesm23nat.zip"
|
|
out = download(client, 2023, dest)
|
|
assert out == dest and zipfile.is_zipfile(dest)
|
|
assert "Mozilla" in seen["ua"] and "text/html" in seen["accept"]
|
|
assert seen["url"] == zip_url(2023)
|
|
|
|
def test_403_raises(self, tmp_path):
|
|
client = httpx.Client(
|
|
transport=httpx.MockTransport(lambda r: httpx.Response(403, text="denied"))
|
|
)
|
|
with pytest.raises(httpx.HTTPStatusError):
|
|
download(client, 2023, tmp_path / "x.zip")
|
|
|
|
def test_non_zip_body_raises(self, tmp_path):
|
|
client = httpx.Client(
|
|
transport=httpx.MockTransport(
|
|
lambda r: httpx.Response(200, text="<html>bot check</html>")
|
|
)
|
|
)
|
|
with pytest.raises(ValueError, match="not a zip"):
|
|
download(client, 2023, tmp_path / "x.zip")
|
|
|
|
|
|
@pytest.fixture
|
|
def con():
|
|
c = duckdb.connect(":memory:")
|
|
yield c
|
|
c.close()
|
|
|
|
|
|
class TestWriteRead:
|
|
def test_write_replaces_year_and_read_filters_occupations(self, con):
|
|
r13 = to_rows(2013, [dict(zip(OLD_COLS, r)) for r in OLD_ROWS], item_key="A")
|
|
r23 = to_rows(2023, [dict(zip(NEW_COLS, r)) for r in NEW_ROWS], item_key="B")
|
|
assert write(con, 2013, r13) == 3 and write(con, 2023, r23) == 2
|
|
assert write(con, 2013, r13) == 3
|
|
assert con.execute("SELECT count(*) FROM bls.oews").fetchone()[0] == 5
|
|
out = read_occupations(con, ["31-9092"])
|
|
assert [(r.year, r.a_mean) for r in out] == [(2013, 30780.0), (2023, 42000.0)]
|
|
assert (
|
|
read_occupations(con, ["31-9092", "29-1141"], years=[2023])[0].occ_code
|
|
== "29-1141"
|
|
)
|
|
|
|
|
|
class TestCite:
|
|
def test_one_source_per_release(self, tmp_path):
|
|
s = Store(":memory:", storage_dir=tmp_path / "st")
|
|
k1, k2, k3 = cite(s, 2023), cite(s, 2023), cite(s, 2013)
|
|
assert k1 == k2 != k3
|
|
item = s.get(k1)
|
|
assert "May 2023" in item.title and item.url == zip_url(2023)
|
|
assert {
|
|
"module:bls",
|
|
"table:bls.oews",
|
|
"source:bls-website",
|
|
"year:2023",
|
|
} <= set(item.tags)
|
|
s.close()
|