diff --git a/notebooks/103_data_preprocess_population_un-owid-2025.py b/notebooks/103_data_preprocess_population_un-owid-2025.py index 6c24f41..ddb10f0 100644 --- a/notebooks/103_data_preprocess_population_un-owid-2025.py +++ b/notebooks/103_data_preprocess_population_un-owid-2025.py @@ -139,18 +139,35 @@ # Read the CSV file historical_df = pd.read_csv(project_root / population_historical_path) -# Filter out rows where Code is NaN (these are typically regional aggregates without ISO codes) -historical_df = historical_df[historical_df["Code"].notna()] +# Keep ISO3 country rows plus the named world key ONLY. OWID ships synthetic +# aggregate codes (OWID_*, income groups) in the same export; today only +# OWID_WRL (wanted) and OWID_KOS are present, but the policy must not depend +# on that staying true -- a future OWID_AFR/OWID_EUR row would otherwise +# enter the intermediate artifact and rely on downstream source +# intersections to be caught. Dropped codes are reported, never silent. +_iso3_mask = historical_df["Code"].astype(str).str.match(r"^[A-Z]{3}$", na=False) +_world_mask = historical_df["Code"] == population_historical_world_key +_dropped_codes = sorted( + set(historical_df.loc[~(_iso3_mask | _world_mask), "Code"].dropna()) +) +if _dropped_codes: + print(f"Dropping {len(_dropped_codes)} non-ISO3 aggregate code(s): {_dropped_codes}") +historical_df = historical_df[_iso3_mask | _world_mask] # Filter the data to start from 1850 historical_df = historical_df[historical_df["Year"] >= 1850] -# Rename columns for consistency with our pipeline +# Rename columns for consistency with our pipeline. The population column +# name is not stable across OWID vintages (pandas rename would silently +# no-op on an absent key, surfacing later as a bare KeyError two lines +# down) -- resolve it through the library helper instead. +from fair_shares.library.iamc_historical.socioeconomic import owid_population_column + historical_df = historical_df.rename( columns={ "Code": "iso3c", "Year": "year", - "Population (historical estimates)": "population", + owid_population_column(historical_df.columns): "population", } ) diff --git a/src/fair_shares/library/iamc_historical/socioeconomic.py b/src/fair_shares/library/iamc_historical/socioeconomic.py index 14c409f..836f606 100644 --- a/src/fair_shares/library/iamc_historical/socioeconomic.py +++ b/src/fair_shares/library/iamc_historical/socioeconomic.py @@ -168,20 +168,40 @@ def _single_unit(scenario: pyam.IamDataFrame, variable: str) -> str: return units[0] + +# OWID rewrites its population export in place with no version identifier, +# so the value column name is not stable -- it has already changed once +# (from "Population (historical estimates)" to "Population"). Newest name +# first; add to the front when it changes again. +_OWID_POPULATION_COLUMNS = ("Population", "Population (historical estimates)") + + +def owid_population_column(columns) -> str: + """Return whichever population column this OWID vintage actually ships.""" + for name in _OWID_POPULATION_COLUMNS: + if name in columns: + return name + raise DataProcessingError( + "No recognised population column in the OWID export. Expected one of " + f"{list(_OWID_POPULATION_COLUMNS)}; found {list(columns)}. OWID rewrites " + "this file in place, so an upstream rename is the likely cause -- add " + "the new name to _OWID_POPULATION_COLUMNS." + ) + + def _load_population(source: SourceLike, *, unit: str) -> pd.DataFrame: if isinstance(source, pd.DataFrame): return source.copy() path = _resolve_path(source, default=_DEFAULT_POPULATION_CSV) long = pd.read_csv(path) long = long[long["Code"].astype(str).str.match(_ISO3_RE, na=False)] + pop_col = owid_population_column(long.columns) wide = ( - long.rename(columns={"Code": "region", "Year": "year"})[ - ["region", "year", "Population (historical estimates)"] - ] + long.rename(columns={"Code": "region", "Year": "year"})[["region", "year", pop_col]] .pivot_table( index="region", columns="year", - values="Population (historical estimates)", + values=pop_col, aggfunc="first", ) .reset_index() diff --git a/tests/unit/iamc_historical/test_owid_population_column.py b/tests/unit/iamc_historical/test_owid_population_column.py new file mode 100644 index 0000000..9be6331 --- /dev/null +++ b/tests/unit/iamc_historical/test_owid_population_column.py @@ -0,0 +1,57 @@ +"""Tests for the OWID population column resolution. + +OWID rewrites its population export in place with no version identifier, and +has already renamed the value column once ("Population (historical +estimates)" -> "Population"). The loader must accept every known spelling and +fail with a diagnostic -- not a bare KeyError -- on an unknown one. A fresh +`fetch-data` pulls whatever OWID currently serves (the source is registered +`unversioned`), so a new user following the README is the most likely person +to hit a rename first. +""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from fair_shares.library.exceptions import DataProcessingError +from fair_shares.library.iamc_historical.socioeconomic import ( + _load_population, + owid_population_column, +) + + +def _owid_csv(tmp_path, population_column: str): + path = tmp_path / "population.csv" + pd.DataFrame( + { + "Entity": ["Brazil", "Brazil", "India", "India"], + "Code": ["BRA", "BRA", "IND", "IND"], + "Year": [2019, 2020, 2019, 2020], + population_column: [211e6, 213e6, 1366e6, 1380e6], + } + ).to_csv(path, index=False) + return path + + +@pytest.mark.parametrize( + "column", ["Population", "Population (historical estimates)"] +) +def test_both_known_owid_spellings_load(tmp_path, column) -> None: + wide = _load_population(_owid_csv(tmp_path, column), unit="million") + + assert set(wide["region"]) == {"bra", "ind"} + bra = wide[wide["region"] == "bra"].iloc[0] + assert bra[2020] == pytest.approx(213.0) # people -> million + + +def test_unknown_column_fails_with_diagnostic(tmp_path) -> None: + with pytest.raises(DataProcessingError, match="Population"): + _load_population( + _owid_csv(tmp_path, "Population (2027 revision)"), unit="million" + ) + + +def test_newest_spelling_wins_when_both_present() -> None: + columns = ["Entity", "Code", "Year", "Population (historical estimates)", "Population"] + assert owid_population_column(columns) == "Population"