Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,12 @@ This project uses [Semantic Versioning](https://semver.org/spec/v2.0.0.html).

## [Unreleased]

## [0.6.0] — 2026-06-16

### Changed

- **`CensusAPI.load()` now recovers its constructor arguments from the dataset name and long data** instead of a stored `_morpc` block. `save()` no longer writes the private `_morpc` descriptor — the resource carries only standard frictionless fields. `survey`/`year`/`variables`/`group` come from the long data; `sumlevel` from the `geoidfq` summary level (with the name token as fallback); `scope` from the canonical dataset name. `load()` no longer raises on resources lacking a `_morpc` block; it raises `ValueError` only when the name/data are inconsistent with a known scope.

## [0.5.0] — 2026-06-16

### Added
Expand Down
131 changes: 107 additions & 24 deletions morpc_census/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -1210,20 +1210,104 @@ def create_resource(self):
'path': self.filename,
'schema': self.schema_filename,
'sources': [{'title': 'US Census Bureau API', 'path': self.request['url'], '_params': self.request['params']}],
# Constructor arguments, captured so CensusAPI.load() can faithfully
# rebuild the instance from this resource without re-fetching.
'_morpc': self._reconstruction_metadata(),
})

def _reconstruction_metadata(self) -> dict:
"""Constructor arguments needed by :meth:`load` to rebuild this instance."""
@staticmethod
def _recover_variable_codes(long: pd.DataFrame) -> set[str]:
"""Reconstruct the full Census variable codes (with type suffix) in *long*.

Inverse of the variable-code splitting done in :meth:`melt`: each value-type
column maps back to its suffix via :data:`VARIABLE_TYPES`, except legacy
decennial codes (which carry no suffix and are emitted verbatim).
"""
inverse_types = {v: k for k, v in VARIABLE_TYPES.items()}
legacy_re = re.compile(r'^[A-Z]+\d{3}[A-Z]?\d{3}$')
codes: set[str] = set()
for vt in (c for c in long.columns if c in VARIABLE_TYPES.values()):
base = long.loc[long[vt].notna(), 'variable']
full = base + inverse_types[vt]
if vt == 'total':
full = full.where(~base.str.match(legacy_re), base)
codes.update(full)
return codes

@staticmethod
def _recover_metadata(name: str, long: pd.DataFrame) -> dict:
"""Recover the constructor arguments for :meth:`load` from saved output.

Reconstructs the six arguments that :meth:`save` does not store explicitly,
using only the canonical dataset *name* (see :func:`censusapi_name`) and the
long-format data:

- ``survey`` / ``year`` — the uniform ``survey`` / ``reference_period`` columns.
- ``variables`` — base ``variable`` codes re-suffixed from the present
value-type columns (see :meth:`_recover_variable_codes`); ``None`` unless
*name* carries the ``-select-variables`` marker.
- ``group`` — table code parsed from the variable codes, kept only
when *name* carries the matching ``-{code}`` segment.
- ``sumlevel`` — summary level parsed from the ``geoidfq`` values, kept
only when *name* carries its hierarchy token (the geoids always encode a
summary level, but the name omits the token when ``sumlevel`` was ``None``).
- ``scope`` — the scope key remaining in *name* once the survey,
year, sumlevel, group, and variable markers are removed.

Raises
------
ValueError
If *name* does not match the survey/year of the data, or the remaining
token is not a recognised scope.
"""
from morpc_census.geos import SCOPES, GeoIDFQ

survey = str(long['survey'].iloc[0])
year = int(long['reference_period'].iloc[0])

# censusapi_name() lowercases everything and lays the parts out as
# census-{survey}-{year}-{sumlevel_part}{scope}{group_part}{var_part}.
prefix = f"census-{survey.replace('/', '-')}-{year}-".lower()
if not name.startswith(prefix):
raise ValueError(
f"Dataset name {name!r} does not match survey/year {survey}/{year}."
)
blob = name[len(prefix):]

# var_part: a trailing '-select-variables' marks an explicit variable list.
has_variables = blob.endswith('-select-variables')
if has_variables:
blob = blob[: -len('-select-variables')]

# group_part: a trailing '-{code}' for a table code that appears in the data.
group = None
group_codes = {_group_code_from_variable(v) for v in long['variable'].unique()}
group_codes.discard('')
for code in group_codes:
if blob.endswith(f"-{code.lower()}"):
group = code
blob = blob[: -(len(code) + 1)]
break

variables = sorted(CensusAPI._recover_variable_codes(long)) if has_variables else None

# sumlevel_part: a leading '{hierarchy-token}-' derived from the geoidfqs.
sl = GeoIDFQ.parse(str(long['geoidfq'].iloc[0])).sumlevel
token = (sl.hierarchy_string or sl.name).replace('-', '').lower()
sumlevel = None
if blob.startswith(f"{token}-") and blob[len(token) + 1:] in SCOPES:
sumlevel = sl.name
blob = blob[len(token) + 1:]

if blob not in SCOPES:
raise ValueError(
f"Could not recover a known scope from dataset name {name!r}; got {blob!r}."
)

return {
'survey': self.endpoint.survey,
'year': self.endpoint.year,
'scope': self.scope.name,
'sumlevel': None if self.sumlevel is None else self.sumlevel.name,
'group': None if self.group is None else self.group.code,
'variables': self.variables,
'survey': survey,
'year': year,
'scope': blob,
'sumlevel': sumlevel,
'group': group,
'variables': variables,
}

def save(self, output_path):
Expand Down Expand Up @@ -1286,6 +1370,10 @@ def load(cls, resource_path) -> "CensusAPI":
``Group`` may still make lightweight metadata lookups, as during normal
construction; the large data fetch is what is skipped.)

The constructor arguments are recovered from the dataset name and the long
data via :meth:`_recover_metadata`; nothing beyond the standard frictionless
resource fields needs to be stored at save time.

Parameters
----------
resource_path : str or path-like
Expand All @@ -1300,9 +1388,9 @@ def load(cls, resource_path) -> "CensusAPI":
------
FileNotFoundError
If the resource file or its referenced long CSV is missing.
RuntimeError
If the resource predates load() support (no ``_morpc`` block) and
therefore lacks the metadata needed to rebuild the instance.
ValueError
If the constructor arguments cannot be recovered from the dataset
name and long data (see :meth:`_recover_metadata`).

Examples
--------
Expand All @@ -1316,26 +1404,21 @@ def load(cls, resource_path) -> "CensusAPI":
raise FileNotFoundError(f"Resource file not found: {resource_path}")

descriptor = frictionless.Resource.from_descriptor(str(resource_path)).to_descriptor()
meta = descriptor.get('_morpc')
if meta is None:
raise RuntimeError(
f"{resource_path} has no '_morpc' metadata block; it was saved by a "
"version predating CensusAPI.load() support and cannot be reconstructed. "
"Re-save it with the current version to enable loading."
)

long_path = resource_path.parent / descriptor['path']
if not long_path.exists():
raise FileNotFoundError(f"Long-format data file not found: {long_path}")
long = pd.read_csv(long_path)

meta = cls._recover_metadata(descriptor['name'], long)

endpoint = Endpoint(meta['survey'], meta['year'])
obj = cls(
endpoint,
meta['scope'],
group=meta.get('group'),
sumlevel=meta.get('sumlevel'),
variables=meta.get('variables'),
group=meta['group'],
sumlevel=meta['sumlevel'],
variables=meta['variables'],
_skip_fetch=True,
)
# Restore the original request from the saved descriptor rather than
Expand Down
30 changes: 30 additions & 0 deletions reference/dev_notes.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,33 @@
## Recover CensusAPI.load() metadata from name + long data (drop _morpc block)

2026-06-16. `load()` no longer depends on the `_morpc` descriptor that #114 added
to the resource; `save()`/`create_resource()` stop writing it. All six
constructor args are recovered instead by the new `_recover_metadata(name, long)`:

- `survey` / `year` — the uniform `survey` / `reference_period` columns.
- `variables` — base `variable` codes re-suffixed from the present value-type
columns via new `_recover_variable_codes()` (the same inverse-`VARIABLE_TYPES`
mapping `_long_to_data()` uses; legacy decennial codes stay verbatim). Returned
only when the name carries the `-select-variables` marker, else `None`.
- `group` — table code from `_group_code_from_variable()` on the variable codes,
kept only when the name ends with the matching `-{code}` segment (this is what
distinguishes group-only / variables-only / both modes — the long data alone
can't, since all three can yield identical rows).
- `sumlevel` — parsed from the `geoidfq` summary level (`GeoIDFQ.parse(...).sumlevel`),
kept only when the name carries that hierarchy token. The geoids always encode
a summary level, so the name's presence/absence of the token is the source of
truth for whether `sumlevel` was supplied.
- `scope` — whatever known `SCOPES` key remains in the name once the survey,
year, sumlevel, group, and variable markers are stripped (anchored parsing:
the other parts are recovered from the data first, so the survey's embedded
dashes don't make the parse ambiguous).

`load()` no longer raises when a `_morpc` block is absent; it raises `ValueError`
only if the name doesn't match the data's survey/year or the residual token isn't
a known scope. Tests: replaced `test_resource_without_morpc_raises` with
`test_resource_has_no_morpc_block` and added recovery tests for all three modes
(group-only/no-sumlevel, variables-only, group+variables). 220 api tests pass.

## Add CensusAPI.load() — reconstruct an instance from save() output (#114)

2026-06-16. Round-trips `save()`: `CensusAPI.load(resource_path)` rebuilds a
Expand Down
60 changes: 53 additions & 7 deletions tests/test_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -2417,11 +2417,57 @@ def test_missing_resource_file_raises(self, tmp_path):
with pytest.raises(FileNotFoundError):
CensusAPI.load(tmp_path / 'does-not-exist.resource.yaml')

def test_resource_without_morpc_raises(self, tmp_path):
def test_resource_has_no_morpc_block(self, tmp_path):
"""save() no longer stores a private _morpc block; the resource carries only
standard frictionless fields and is still loadable."""
import frictionless
(tmp_path / 'x.long.csv').write_text('geoidfq,variable,estimate\n0400000US39,B01001_001,1\n')
desc = {'name': 'x', 'path': 'x.long.csv',
'sources': [{'title': 'US Census Bureau API', 'path': 'http://x'}]}
frictionless.Resource.from_descriptor(desc).to_yaml(str(tmp_path / 'x.resource.yaml'))
with pytest.raises(RuntimeError, match='_morpc'):
CensusAPI.load(tmp_path / 'x.resource.yaml')
api = self._saved_api(tmp_path)
descriptor = frictionless.Resource.from_descriptor(
str(tmp_path / f'{api.name}.resource.yaml')
).to_descriptor()
assert '_morpc' not in descriptor
CensusAPI.load(tmp_path / f'{api.name}.resource.yaml')

# -- recovery across the three group/variables modes -------------------

def _save_and_load(self, tmp_path, **kwargs):
api = CensusAPI(Endpoint('acs/acs5', 2023), 'franklin', _skip_fetch=True, **kwargs)
api.request = {
'url': 'https://api.census.gov/data/2023/acs/acs5?',
'params': {'get': 'group(B01001)', 'for': 'county:049,041', 'in': 'state:39'},
}
api.long = self._make_long()
api.save(tmp_path)
return api, CensusAPI.load(tmp_path / f'{api.name}.resource.yaml')

def test_recover_group_only_no_sumlevel(self, tmp_path):
api, loaded = self._save_and_load(tmp_path, group='B01001')
assert loaded.name == api.name
assert loaded.scope.name == 'franklin'
assert loaded.sumlevel is None
assert loaded.group.code == 'B01001'
assert loaded.variables is None

def test_recover_variables_only(self, tmp_path):
variables = ['B01001_001E', 'B01001_001M', 'B01001_002E', 'B01001_002M']
api, loaded = self._save_and_load(tmp_path, variables=variables, sumlevel='county')
assert loaded.name == api.name
assert loaded.scope.name == 'franklin'
assert loaded.sumlevel.name == 'county'
assert loaded.group is None
assert set(loaded.variables) == set(variables)

def test_recover_group_and_variables(self, tmp_path):
# The requested subset must match what _make_long() carries (E + M for both
# base codes), so the round-trip recovers exactly these variables.
variables = ['B01001_001E', 'B01001_001M', 'B01001_002E', 'B01001_002M']
group_vars = {v: {} for v in variables}
with patch.object(Group, 'variables', property(lambda self: group_vars)):
api, loaded = self._save_and_load(
tmp_path, group='B01001', variables=variables, sumlevel='county',
)
assert loaded.name == api.name
assert loaded.scope.name == 'franklin'
assert loaded.sumlevel.name == 'county'
assert loaded.group.code == 'B01001'
assert set(loaded.variables) == set(variables)
Loading