Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,12 @@ This project uses [Semantic Versioning](https://semver.org/spec/v2.0.0.html).

## [Unreleased]

## [0.6.3] — 2026-09-23

### Fixed

- **`CensusAPI` for sumlevels that don't nest in the scope** (e.g. place-county parts, `155`, in `region15`) no longer fails with HTTP 400 (previously surfaced as `UnboundLocalError: json`). `geoinfo_for_hierarchical_geos()` now requests one geography per query, finds the places that intersect the scope with a single pseudo query instead of every place in the state, removes duplicate parents, and keeps only results inside the scope. The place list comes from current geography, so parts of places that no longer exist (e.g. some 2000/2010 CDPs) are not returned.

## [0.6.2] — 2026-09-23

### Fixed
Expand Down
51 changes: 38 additions & 13 deletions morpc_census/geos.py
Original file line number Diff line number Diff line change
Expand Up @@ -548,7 +548,12 @@ def pseudos_from_scope_sumlevel(


def geoinfo_for_hierarchical_geos(scope: str | Scope, sumlevel: str | SumLevel) -> DataFrame:
"""Build a geoinfo table for sumlevel/scope combinations that cannot be expressed as ucgid pseudos."""
"""Build a geoinfo table for sumlevel/scope combinations that cannot be expressed as ucgid pseudos.

The Census API requires some geographies (e.g. ``place`` for place-county parts) that do not nest in the
scope. Each is resolved to the values that intersect the scope, the sumlevel is queried once per
combination, and only results inside the scope are kept (e.g. the county parts in the scope's counties).
"""
import pandas as pd
sl = sumlevel if isinstance(sumlevel, SumLevel) else SumLevel(sumlevel)
sc = scope if isinstance(scope, Scope) else SCOPES[scope]
Expand All @@ -558,25 +563,37 @@ def geoinfo_for_hierarchical_geos(scope: str | Scope, sumlevel: str | SumLevel)
in_scope = [x.split(":")[0] for x in scope_params if x.split(":")[0] in query_req['requires']]
not_in_scope = [x for x in query_req['requires'] if x not in in_scope]

parent_table = geoids_from_scope(sc, output='table').drop(columns='GEO_ID')
scope_table = geoids_from_scope(sc, output='table')
parent_table = scope_table.drop(columns='GEO_ID')

for geo in not_in_scope:
logger.info(f"Getting {geo} from {in_scope}")
parent_table[geo] = None
parent_table[geo] = parent_table[geo].astype('object')

for i, row in parent_table.iterrows():
in_param_str = [
f"{x}:{','.join(row[x]) if isinstance(row[x], list) else row[x]}"
for x in in_scope
]
geoids = geoinfo_from_params({"in": in_param_str, "for": f"{geo}:*"}, output='table')[geo].to_list()
logger.debug(f"at row:{i}, column:{geo} adding geoids {geoids}")
parent_table.at[i, geo] = geoids
try:
# One pseudo query finds only the geographies of this type that intersect the scope.
pseudos = pseudos_from_scope_sumlevel(SumLevel(geo), sc)
found = geoinfo_from_params({'ucgid': f"pseudo({','.join(pseudos)})"}, output='table')
geoids = sorted({getattr(GeoIDFQ.parse(fq), geo) for fq in found['GEO_ID']})
for i in parent_table.index:
parent_table.at[i, geo] = geoids
except (ValueError, KeyError):
for i, row in parent_table.iterrows():
in_param_str = [
f"{x}:{','.join(row[x]) if isinstance(row[x], list) else row[x]}"
for x in in_scope
]
geoids = geoinfo_from_params({"in": in_param_str, "for": f"{geo}:*"}, output='table')[geo].to_list()
logger.debug(f"at row:{i}, column:{geo} adding geoids {geoids}")
parent_table.at[i, geo] = geoids

in_scope += [geo]
if len(in_scope) < len(query_req['requires']):
parent_table = parent_table.explode(geo).reset_index().drop(columns='index')
# One row per value, so every request names a single geography. The API rejects a list for most of them.
parent_table = parent_table.explode(geo).dropna(subset=[geo]).reset_index().drop(columns='index')

# The same parent combination can come from several scope rows (e.g. a place in two scope counties).
parent_table = parent_table[in_scope].drop_duplicates()

geoinfos = []
for i, row in parent_table.iterrows():
Expand All @@ -586,8 +603,16 @@ def geoinfo_for_hierarchical_geos(scope: str | Scope, sumlevel: str | SumLevel)
]
geoinfo = geoinfo_from_params({"in": in_param_str, "for": f"{sl.name}:*"}, output='table')
geoinfos.append(geoinfo)
geoinfo = pd.concat(geoinfos)

# Keep only results inside the scope, e.g. drop the parts of a scope place that lie in other counties.
scope_fields = [f for f in scope_table.columns if f in sl.parts and f not in query_req['requires']]
if scope_fields:
scope_keys = set(scope_table[scope_fields].itertuples(index=False, name=None))
in_scope_rows = geoinfo['GEO_ID'].map(lambda fq: tuple(getattr(GeoIDFQ.parse(fq), f) for f in scope_fields) in scope_keys)
geoinfo = geoinfo[in_scope_rows]

return pd.concat(geoinfos)
return geoinfo


def geoinfo_from_scope_sumlevel(
Expand Down
15 changes: 15 additions & 0 deletions reference/dev_notes.md
Original file line number Diff line number Diff line change
Expand Up @@ -1077,3 +1077,18 @@ Tests: `TestDimensionTableDescriptionTable` replaced by `TestDimensionTableParse
**Root cause**: The implementation transposed `wide()`, dropped the total column, called `reset_index()`, then tried to identify metadata columns via `non_value_cols = [c for c in pct.columns if c not in self.variable_type]`. After `reset_index()` on a transposed DataFrame, the columns were MultiIndex tuples like `('Total', 'Male')` and `('Total', 'Female')`. These tuples are not in `variable_type` (`['estimate', 'moe']`), so they were included in `non_value_cols` and became index levels. The subsequent `.T` then put them into column headers.

**Fix**: Replaced the transpose-and-reconstruct approach with a direct operation on the `wide()` output — find the total row by integer position, divide each column individually by its total value, drop the total row, and return. Output has the same structure as `wide()` (dimension values as row index, geographies as column MultiIndex) but with percentage values and no total row.

## 2026-09-23 — Fix geoinfo_for_hierarchical_geos() for place-county parts (branch fix/hierarchical-place-parts)

**Bug**: `CensusAPI(Endpoint('dec/pl', 2020), 'region15', sumlevel=SumLevel('155'), ...)` failed with `UnboundLocalError: json` (an HTTP 400 hidden by `morpc.req.get_json_safely`, fixed in morpc 0.7.3).

**Root cause**: 155 requires `state` and `place`, and `place` does not nest under county, so pseudos are unavailable and the hierarchical fallback runs. It had three problems:
1. It exploded a resolved geography only while more requirements remained, so the last one (`place`) stayed a list and the final request sent all 1,265 Ohio places comma-joined in one `in=` clause. The API allows a single place → 400.
2. For each of the 15 scope counties it fetched every place in the state (`for=place:*&in=state:39`), since that hierarchy cannot filter by county: 15 duplicate lists, and after fixing (1) about 19,000 requests.
3. Nothing limited results to the scope, so parts of scope places lying in other counties would be returned.

**Fix**: For each missing requirement, try one pseudo query from the scope (e.g. `050$1600000`, 191 places for region15) before falling back to the per-row for/in lookup. Always explode, drop duplicate parent combinations, and filter results on the scope's fields that are part of the sumlevel's GEOIDFQ but not required by the API (e.g. `county`). Region15 155 now takes ~191 requests per call (~100 s).

**Limitation**: the place list comes from the 2024 geoinfo, so places that no longer exist (e.g. Hidden Lakes CDP, 2020) are not returned.

Tests: `tests/test_geos_hierarchical.py` — 3 new tests (scope filtering, one request per place, fallback path). 340 passing. Verified live for 2000/2010/2020 dec/pl: Columbus parts sum to the place total (905,748 in 2020); every other part/place mismatch but Hidden Lakes is an out-of-region county part.
78 changes: 78 additions & 0 deletions tests/test_geos_hierarchical.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
from unittest.mock import patch

import pandas as pd
import pytest

from morpc_census.geos import Scope, SumLevel, geoinfo_for_hierarchical_geos

# Place-county parts (155) require state and place, but place does not nest under county, so a county-based
# scope must find the places that intersect it, query each place's county parts, and keep the in-scope ones.

SCOPE = Scope(name="twocounty", for_param="county:041,049", in_param="state:39")

SCOPE_TABLE = pd.DataFrame({
"GEO_ID": ["0500000US39041", "0500000US39049"],
"state": ["39", "39"],
"county": ["041", "049"],
})

# Columbus (18000) spans Delaware (041), Fairfield (045), and Franklin (049); Ashley (02582) is in Delaware only.
# 99999 is a place elsewhere in Ohio, which only the fallback's statewide place list includes.
PARTS = {"18000": ["041", "045", "049"], "02582": ["041"], "99999": ["003"]}


def _parts_table(place):
geoids = [f"1550000US39{place}{county}" for county in PARTS[place]]
return pd.DataFrame({"GEO_ID": geoids, "NAME": geoids})


def _fake_geoinfo(calls):
def fake(param_dict, *args, **kwargs):
calls.append(param_dict)
if "ucgid" in param_dict:
# A pseudo query lists a place once per county it intersects.
geoids = ["1600000US3918000", "1600000US3918000", "1600000US3902582"]
return pd.DataFrame({"GEO_ID": geoids, "NAME": geoids, "ucgid": geoids})
if param_dict["for"] == "place:*":
return pd.DataFrame({"place": ["18000", "02582", "99999"]})
place = [p for p in param_dict["in"] if p.startswith("place:")][0].split(":")[1]
return _parts_table(place)
return fake


def _run(calls, pseudo_available=True):
pseudo = (patch("morpc_census.geos.pseudos_from_scope_sumlevel", return_value=["x"]) if pseudo_available
else patch("morpc_census.geos.pseudos_from_scope_sumlevel", side_effect=ValueError))
with patch.object(SumLevel, "get_query_req", return_value={"requires": ["state", "place"], "wildcard": None}), \
patch("morpc_census.geos.geoids_from_scope", return_value=SCOPE_TABLE), \
patch("morpc_census.geos.geoinfo_from_params", side_effect=_fake_geoinfo(calls)), \
pseudo:
return geoinfo_for_hierarchical_geos(SCOPE, SumLevel("155"))


def _part_requests(calls):
return [c for c in calls if c.get("for") == "county (or part):*"]


def test_place_county_parts_are_limited_to_scope_counties():
calls = []
result = _run(calls)
# Fairfield (045) is outside the scope, so its part of Columbus is dropped.
assert sorted(result["GEO_ID"]) == ["1550000US3902582041", "1550000US3918000041", "1550000US3918000049"]


def test_each_place_is_queried_once_with_a_single_place():
calls = []
_run(calls)
places = [[p for p in c["in"] if p.startswith("place:")][0] for c in _part_requests(calls)]
# The API rejects a list of places, and a place spanning two scope counties must not be queried twice.
assert sorted(places) == ["place:02582", "place:18000"]


def test_fallback_without_pseudo_also_queries_single_places():
calls = []
result = _run(calls, pseudo_available=False)
places = [[p for p in c["in"] if p.startswith("place:")][0] for c in _part_requests(calls)]
assert all("," not in p for p in places)
assert len(places) == len(set(places))
assert sorted(result["GEO_ID"]) == ["1550000US3902582041", "1550000US3918000041", "1550000US3918000049"]
Loading