diff --git a/src/bls/oews.py b/src/bls/oews.py index 5a560d3..1c2e87a 100644 --- a/src/bls/oews.py +++ b/src/bls/oews.py @@ -129,7 +129,8 @@ def read_national(zip_path: Path, year: int) -> list[dict]: if ext == "xlsx" else pl.read_excel(io.BytesIO(raw), infer_schema_length=0, engine="calamine") ) - return df.to_dicts() + # The 2019 release shipped lower-case headers; every other year is upper. + return df.rename({c: c.upper() for c in df.columns}).to_dicts() # ── rows ────────────────────────────────────────────────────────────── diff --git a/tests/bls/test_oews.py b/tests/bls/test_oews.py index de40d2f..8f9dd2e 100644 --- a/tests/bls/test_oews.py +++ b/tests/bls/test_oews.py @@ -271,6 +271,14 @@ class TestReadNational: recs = read_national(p, 2023) assert len(recs) == 2 and recs[0]["O_GROUP"] == "detailed" + def test_lower_case_headers_are_normalised(self, tmp_path): + # the May 2019 file is the one release with lower-case column names + p = tmp_path / "oesm19nat.zip" + p.write_bytes(_zip(2019, [c.lower() for c in NEW_COLS], NEW_ROWS)) + recs = read_national(p, 2019) + assert recs[0]["OCC_CODE"] == "31-9092" + assert len(to_rows(2019, recs, item_key="K")) == 2 + def test_missing_member(self, tmp_path): p = tmp_path / "bad.zip" with zipfile.ZipFile(p, "w") as z: