Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 21 additions & 4 deletions notebooks/103_data_preprocess_population_un-owid-2025.py
Original file line number Diff line number Diff line change
Expand Up @@ -139,18 +139,35 @@
# Read the CSV file
historical_df = pd.read_csv(project_root / population_historical_path)

# Filter out rows where Code is NaN (these are typically regional aggregates without ISO codes)
historical_df = historical_df[historical_df["Code"].notna()]
# Keep ISO3 country rows plus the named world key ONLY. OWID ships synthetic
# aggregate codes (OWID_*, income groups) in the same export; today only
# OWID_WRL (wanted) and OWID_KOS are present, but the policy must not depend
# on that staying true -- a future OWID_AFR/OWID_EUR row would otherwise
# enter the intermediate artifact and rely on downstream source
# intersections to be caught. Dropped codes are reported, never silent.
_iso3_mask = historical_df["Code"].astype(str).str.match(r"^[A-Z]{3}$", na=False)
_world_mask = historical_df["Code"] == population_historical_world_key
_dropped_codes = sorted(
set(historical_df.loc[~(_iso3_mask | _world_mask), "Code"].dropna())
)
if _dropped_codes:
print(f"Dropping {len(_dropped_codes)} non-ISO3 aggregate code(s): {_dropped_codes}")
historical_df = historical_df[_iso3_mask | _world_mask]

# Filter the data to start from 1850
historical_df = historical_df[historical_df["Year"] >= 1850]

# Rename columns for consistency with our pipeline
# Rename columns for consistency with our pipeline. The population column
# name is not stable across OWID vintages (pandas rename would silently
# no-op on an absent key, surfacing later as a bare KeyError two lines
# down) -- resolve it through the library helper instead.
from fair_shares.library.iamc_historical.socioeconomic import owid_population_column

historical_df = historical_df.rename(
columns={
"Code": "iso3c",
"Year": "year",
"Population (historical estimates)": "population",
owid_population_column(historical_df.columns): "population",
}
)

Expand Down
28 changes: 24 additions & 4 deletions src/fair_shares/library/iamc_historical/socioeconomic.py
Original file line number Diff line number Diff line change
Expand Up @@ -168,20 +168,40 @@ def _single_unit(scenario: pyam.IamDataFrame, variable: str) -> str:
return units[0]



# OWID rewrites its population export in place with no version identifier,
# so the value column name is not stable -- it has already changed once
# (from "Population (historical estimates)" to "Population"). Newest name
# first; add to the front when it changes again.
_OWID_POPULATION_COLUMNS = ("Population", "Population (historical estimates)")


def owid_population_column(columns) -> str:
"""Return whichever population column this OWID vintage actually ships."""
for name in _OWID_POPULATION_COLUMNS:
if name in columns:
return name
raise DataProcessingError(
"No recognised population column in the OWID export. Expected one of "
f"{list(_OWID_POPULATION_COLUMNS)}; found {list(columns)}. OWID rewrites "
"this file in place, so an upstream rename is the likely cause -- add "
"the new name to _OWID_POPULATION_COLUMNS."
)


def _load_population(source: SourceLike, *, unit: str) -> pd.DataFrame:
if isinstance(source, pd.DataFrame):
return source.copy()
path = _resolve_path(source, default=_DEFAULT_POPULATION_CSV)
long = pd.read_csv(path)
long = long[long["Code"].astype(str).str.match(_ISO3_RE, na=False)]
pop_col = owid_population_column(long.columns)
wide = (
long.rename(columns={"Code": "region", "Year": "year"})[
["region", "year", "Population (historical estimates)"]
]
long.rename(columns={"Code": "region", "Year": "year"})[["region", "year", pop_col]]
.pivot_table(
index="region",
columns="year",
values="Population (historical estimates)",
values=pop_col,
aggfunc="first",
)
.reset_index()
Expand Down
57 changes: 57 additions & 0 deletions tests/unit/iamc_historical/test_owid_population_column.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,57 @@
"""Tests for the OWID population column resolution.

OWID rewrites its population export in place with no version identifier, and
has already renamed the value column once ("Population (historical
estimates)" -> "Population"). The loader must accept every known spelling and
fail with a diagnostic -- not a bare KeyError -- on an unknown one. A fresh
`fetch-data` pulls whatever OWID currently serves (the source is registered
`unversioned`), so a new user following the README is the most likely person
to hit a rename first.
"""

from __future__ import annotations

import pandas as pd
import pytest

from fair_shares.library.exceptions import DataProcessingError
from fair_shares.library.iamc_historical.socioeconomic import (
_load_population,
owid_population_column,
)


def _owid_csv(tmp_path, population_column: str):
path = tmp_path / "population.csv"
pd.DataFrame(
{
"Entity": ["Brazil", "Brazil", "India", "India"],
"Code": ["BRA", "BRA", "IND", "IND"],
"Year": [2019, 2020, 2019, 2020],
population_column: [211e6, 213e6, 1366e6, 1380e6],
}
).to_csv(path, index=False)
return path


@pytest.mark.parametrize(
"column", ["Population", "Population (historical estimates)"]
)
def test_both_known_owid_spellings_load(tmp_path, column) -> None:
wide = _load_population(_owid_csv(tmp_path, column), unit="million")

assert set(wide["region"]) == {"bra", "ind"}
bra = wide[wide["region"] == "bra"].iloc[0]
assert bra[2020] == pytest.approx(213.0) # people -> million


def test_unknown_column_fails_with_diagnostic(tmp_path) -> None:
with pytest.raises(DataProcessingError, match="Population"):
_load_population(
_owid_csv(tmp_path, "Population (2027 revision)"), unit="million"
)


def test_newest_spelling_wins_when_both_present() -> None:
columns = ["Entity", "Code", "Year", "Population (historical estimates)", "Population"]
assert owid_population_column(columns) == "Population"
Loading