fix(bls): upper-case the OEWS sheet headers — the May 2019 release ships them lower-case, which loaded 0 occupations (refs #695)
All checks were successful
CI / lint (push) Successful in 53s
CI / notebooks-smoke (push) Successful in 1m38s
Deploy / notebooks (push) Has been skipped
CI / test (push) Successful in 2m37s
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 14s
Infra CI / notebooks (push) Successful in 53s
Infra CI / docs (push) Successful in 1m31s
Infra CI / api (push) Successful in 1m0s
Infra CI / mc (push) Successful in 17s
Deploy / report (push) Successful in 15s
Infra CI / llm (push) Successful in 51s
All checks were successful
CI / lint (push) Successful in 53s
CI / notebooks-smoke (push) Successful in 1m38s
Deploy / notebooks (push) Has been skipped
CI / test (push) Successful in 2m37s
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Successful in 14s
Infra CI / notebooks (push) Successful in 53s
Infra CI / docs (push) Successful in 1m31s
Infra CI / api (push) Successful in 1m0s
Infra CI / mc (push) Successful in 17s
Deploy / report (push) Successful in 15s
Infra CI / llm (push) Successful in 51s
This commit is contained in:
@@ -129,7 +129,8 @@ def read_national(zip_path: Path, year: int) -> list[dict]:
|
||||
if ext == "xlsx"
|
||||
else pl.read_excel(io.BytesIO(raw), infer_schema_length=0, engine="calamine")
|
||||
)
|
||||
return df.to_dicts()
|
||||
# The 2019 release shipped lower-case headers; every other year is upper.
|
||||
return df.rename({c: c.upper() for c in df.columns}).to_dicts()
|
||||
|
||||
|
||||
# ── rows ──────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -271,6 +271,14 @@ class TestReadNational:
|
||||
recs = read_national(p, 2023)
|
||||
assert len(recs) == 2 and recs[0]["O_GROUP"] == "detailed"
|
||||
|
||||
def test_lower_case_headers_are_normalised(self, tmp_path):
|
||||
# the May 2019 file is the one release with lower-case column names
|
||||
p = tmp_path / "oesm19nat.zip"
|
||||
p.write_bytes(_zip(2019, [c.lower() for c in NEW_COLS], NEW_ROWS))
|
||||
recs = read_national(p, 2019)
|
||||
assert recs[0]["OCC_CODE"] == "31-9092"
|
||||
assert len(to_rows(2019, recs, item_key="K")) == 2
|
||||
|
||||
def test_missing_member(self, tmp_path):
|
||||
p = tmp_path / "bad.zip"
|
||||
with zipfile.ZipFile(p, "w") as z:
|
||||
|
||||
Reference in New Issue
Block a user