diff --git a/CHANGELOG.md b/CHANGELOG.md index 7ac6abe..ded9cd7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,12 @@ This project uses [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +## [0.6.0] — 2026-06-16 + +### Changed + +- **`CensusAPI.load()` now recovers its constructor arguments from the dataset name and long data** instead of a stored `_morpc` block. `save()` no longer writes the private `_morpc` descriptor — the resource carries only standard frictionless fields. `survey`/`year`/`variables`/`group` come from the long data; `sumlevel` from the `geoidfq` summary level (with the name token as fallback); `scope` from the canonical dataset name. `load()` no longer raises on resources lacking a `_morpc` block; it raises `ValueError` only when the name/data are inconsistent with a known scope. + ## [0.5.0] — 2026-06-16 ### Added diff --git a/morpc_census/api.py b/morpc_census/api.py index 669215d..c13f9e5 100644 --- a/morpc_census/api.py +++ b/morpc_census/api.py @@ -1210,20 +1210,104 @@ def create_resource(self): 'path': self.filename, 'schema': self.schema_filename, 'sources': [{'title': 'US Census Bureau API', 'path': self.request['url'], '_params': self.request['params']}], - # Constructor arguments, captured so CensusAPI.load() can faithfully - # rebuild the instance from this resource without re-fetching. - '_morpc': self._reconstruction_metadata(), }) - def _reconstruction_metadata(self) -> dict: - """Constructor arguments needed by :meth:`load` to rebuild this instance.""" + @staticmethod + def _recover_variable_codes(long: pd.DataFrame) -> set[str]: + """Reconstruct the full Census variable codes (with type suffix) in *long*. + + Inverse of the variable-code splitting done in :meth:`melt`: each value-type + column maps back to its suffix via :data:`VARIABLE_TYPES`, except legacy + decennial codes (which carry no suffix and are emitted verbatim). + """ + inverse_types = {v: k for k, v in VARIABLE_TYPES.items()} + legacy_re = re.compile(r'^[A-Z]+\d{3}[A-Z]?\d{3}$') + codes: set[str] = set() + for vt in (c for c in long.columns if c in VARIABLE_TYPES.values()): + base = long.loc[long[vt].notna(), 'variable'] + full = base + inverse_types[vt] + if vt == 'total': + full = full.where(~base.str.match(legacy_re), base) + codes.update(full) + return codes + + @staticmethod + def _recover_metadata(name: str, long: pd.DataFrame) -> dict: + """Recover the constructor arguments for :meth:`load` from saved output. + + Reconstructs the six arguments that :meth:`save` does not store explicitly, + using only the canonical dataset *name* (see :func:`censusapi_name`) and the + long-format data: + + - ``survey`` / ``year`` — the uniform ``survey`` / ``reference_period`` columns. + - ``variables`` — base ``variable`` codes re-suffixed from the present + value-type columns (see :meth:`_recover_variable_codes`); ``None`` unless + *name* carries the ``-select-variables`` marker. + - ``group`` — table code parsed from the variable codes, kept only + when *name* carries the matching ``-{code}`` segment. + - ``sumlevel`` — summary level parsed from the ``geoidfq`` values, kept + only when *name* carries its hierarchy token (the geoids always encode a + summary level, but the name omits the token when ``sumlevel`` was ``None``). + - ``scope`` — the scope key remaining in *name* once the survey, + year, sumlevel, group, and variable markers are removed. + + Raises + ------ + ValueError + If *name* does not match the survey/year of the data, or the remaining + token is not a recognised scope. + """ + from morpc_census.geos import SCOPES, GeoIDFQ + + survey = str(long['survey'].iloc[0]) + year = int(long['reference_period'].iloc[0]) + + # censusapi_name() lowercases everything and lays the parts out as + # census-{survey}-{year}-{sumlevel_part}{scope}{group_part}{var_part}. + prefix = f"census-{survey.replace('/', '-')}-{year}-".lower() + if not name.startswith(prefix): + raise ValueError( + f"Dataset name {name!r} does not match survey/year {survey}/{year}." + ) + blob = name[len(prefix):] + + # var_part: a trailing '-select-variables' marks an explicit variable list. + has_variables = blob.endswith('-select-variables') + if has_variables: + blob = blob[: -len('-select-variables')] + + # group_part: a trailing '-{code}' for a table code that appears in the data. + group = None + group_codes = {_group_code_from_variable(v) for v in long['variable'].unique()} + group_codes.discard('') + for code in group_codes: + if blob.endswith(f"-{code.lower()}"): + group = code + blob = blob[: -(len(code) + 1)] + break + + variables = sorted(CensusAPI._recover_variable_codes(long)) if has_variables else None + + # sumlevel_part: a leading '{hierarchy-token}-' derived from the geoidfqs. + sl = GeoIDFQ.parse(str(long['geoidfq'].iloc[0])).sumlevel + token = (sl.hierarchy_string or sl.name).replace('-', '').lower() + sumlevel = None + if blob.startswith(f"{token}-") and blob[len(token) + 1:] in SCOPES: + sumlevel = sl.name + blob = blob[len(token) + 1:] + + if blob not in SCOPES: + raise ValueError( + f"Could not recover a known scope from dataset name {name!r}; got {blob!r}." + ) + return { - 'survey': self.endpoint.survey, - 'year': self.endpoint.year, - 'scope': self.scope.name, - 'sumlevel': None if self.sumlevel is None else self.sumlevel.name, - 'group': None if self.group is None else self.group.code, - 'variables': self.variables, + 'survey': survey, + 'year': year, + 'scope': blob, + 'sumlevel': sumlevel, + 'group': group, + 'variables': variables, } def save(self, output_path): @@ -1286,6 +1370,10 @@ def load(cls, resource_path) -> "CensusAPI": ``Group`` may still make lightweight metadata lookups, as during normal construction; the large data fetch is what is skipped.) + The constructor arguments are recovered from the dataset name and the long + data via :meth:`_recover_metadata`; nothing beyond the standard frictionless + resource fields needs to be stored at save time. + Parameters ---------- resource_path : str or path-like @@ -1300,9 +1388,9 @@ def load(cls, resource_path) -> "CensusAPI": ------ FileNotFoundError If the resource file or its referenced long CSV is missing. - RuntimeError - If the resource predates load() support (no ``_morpc`` block) and - therefore lacks the metadata needed to rebuild the instance. + ValueError + If the constructor arguments cannot be recovered from the dataset + name and long data (see :meth:`_recover_metadata`). Examples -------- @@ -1316,26 +1404,21 @@ def load(cls, resource_path) -> "CensusAPI": raise FileNotFoundError(f"Resource file not found: {resource_path}") descriptor = frictionless.Resource.from_descriptor(str(resource_path)).to_descriptor() - meta = descriptor.get('_morpc') - if meta is None: - raise RuntimeError( - f"{resource_path} has no '_morpc' metadata block; it was saved by a " - "version predating CensusAPI.load() support and cannot be reconstructed. " - "Re-save it with the current version to enable loading." - ) long_path = resource_path.parent / descriptor['path'] if not long_path.exists(): raise FileNotFoundError(f"Long-format data file not found: {long_path}") long = pd.read_csv(long_path) + meta = cls._recover_metadata(descriptor['name'], long) + endpoint = Endpoint(meta['survey'], meta['year']) obj = cls( endpoint, meta['scope'], - group=meta.get('group'), - sumlevel=meta.get('sumlevel'), - variables=meta.get('variables'), + group=meta['group'], + sumlevel=meta['sumlevel'], + variables=meta['variables'], _skip_fetch=True, ) # Restore the original request from the saved descriptor rather than diff --git a/reference/dev_notes.md b/reference/dev_notes.md index 3fb8622..7d6a0a8 100644 --- a/reference/dev_notes.md +++ b/reference/dev_notes.md @@ -1,3 +1,33 @@ +## Recover CensusAPI.load() metadata from name + long data (drop _morpc block) + +2026-06-16. `load()` no longer depends on the `_morpc` descriptor that #114 added +to the resource; `save()`/`create_resource()` stop writing it. All six +constructor args are recovered instead by the new `_recover_metadata(name, long)`: + +- `survey` / `year` — the uniform `survey` / `reference_period` columns. +- `variables` — base `variable` codes re-suffixed from the present value-type + columns via new `_recover_variable_codes()` (the same inverse-`VARIABLE_TYPES` + mapping `_long_to_data()` uses; legacy decennial codes stay verbatim). Returned + only when the name carries the `-select-variables` marker, else `None`. +- `group` — table code from `_group_code_from_variable()` on the variable codes, + kept only when the name ends with the matching `-{code}` segment (this is what + distinguishes group-only / variables-only / both modes — the long data alone + can't, since all three can yield identical rows). +- `sumlevel` — parsed from the `geoidfq` summary level (`GeoIDFQ.parse(...).sumlevel`), + kept only when the name carries that hierarchy token. The geoids always encode + a summary level, so the name's presence/absence of the token is the source of + truth for whether `sumlevel` was supplied. +- `scope` — whatever known `SCOPES` key remains in the name once the survey, + year, sumlevel, group, and variable markers are stripped (anchored parsing: + the other parts are recovered from the data first, so the survey's embedded + dashes don't make the parse ambiguous). + +`load()` no longer raises when a `_morpc` block is absent; it raises `ValueError` +only if the name doesn't match the data's survey/year or the residual token isn't +a known scope. Tests: replaced `test_resource_without_morpc_raises` with +`test_resource_has_no_morpc_block` and added recovery tests for all three modes +(group-only/no-sumlevel, variables-only, group+variables). 220 api tests pass. + ## Add CensusAPI.load() — reconstruct an instance from save() output (#114) 2026-06-16. Round-trips `save()`: `CensusAPI.load(resource_path)` rebuilds a diff --git a/tests/test_api.py b/tests/test_api.py index aa1189d..193f2e6 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -2417,11 +2417,57 @@ def test_missing_resource_file_raises(self, tmp_path): with pytest.raises(FileNotFoundError): CensusAPI.load(tmp_path / 'does-not-exist.resource.yaml') - def test_resource_without_morpc_raises(self, tmp_path): + def test_resource_has_no_morpc_block(self, tmp_path): + """save() no longer stores a private _morpc block; the resource carries only + standard frictionless fields and is still loadable.""" import frictionless - (tmp_path / 'x.long.csv').write_text('geoidfq,variable,estimate\n0400000US39,B01001_001,1\n') - desc = {'name': 'x', 'path': 'x.long.csv', - 'sources': [{'title': 'US Census Bureau API', 'path': 'http://x'}]} - frictionless.Resource.from_descriptor(desc).to_yaml(str(tmp_path / 'x.resource.yaml')) - with pytest.raises(RuntimeError, match='_morpc'): - CensusAPI.load(tmp_path / 'x.resource.yaml') + api = self._saved_api(tmp_path) + descriptor = frictionless.Resource.from_descriptor( + str(tmp_path / f'{api.name}.resource.yaml') + ).to_descriptor() + assert '_morpc' not in descriptor + CensusAPI.load(tmp_path / f'{api.name}.resource.yaml') + + # -- recovery across the three group/variables modes ------------------- + + def _save_and_load(self, tmp_path, **kwargs): + api = CensusAPI(Endpoint('acs/acs5', 2023), 'franklin', _skip_fetch=True, **kwargs) + api.request = { + 'url': 'https://api.census.gov/data/2023/acs/acs5?', + 'params': {'get': 'group(B01001)', 'for': 'county:049,041', 'in': 'state:39'}, + } + api.long = self._make_long() + api.save(tmp_path) + return api, CensusAPI.load(tmp_path / f'{api.name}.resource.yaml') + + def test_recover_group_only_no_sumlevel(self, tmp_path): + api, loaded = self._save_and_load(tmp_path, group='B01001') + assert loaded.name == api.name + assert loaded.scope.name == 'franklin' + assert loaded.sumlevel is None + assert loaded.group.code == 'B01001' + assert loaded.variables is None + + def test_recover_variables_only(self, tmp_path): + variables = ['B01001_001E', 'B01001_001M', 'B01001_002E', 'B01001_002M'] + api, loaded = self._save_and_load(tmp_path, variables=variables, sumlevel='county') + assert loaded.name == api.name + assert loaded.scope.name == 'franklin' + assert loaded.sumlevel.name == 'county' + assert loaded.group is None + assert set(loaded.variables) == set(variables) + + def test_recover_group_and_variables(self, tmp_path): + # The requested subset must match what _make_long() carries (E + M for both + # base codes), so the round-trip recovers exactly these variables. + variables = ['B01001_001E', 'B01001_001M', 'B01001_002E', 'B01001_002M'] + group_vars = {v: {} for v in variables} + with patch.object(Group, 'variables', property(lambda self: group_vars)): + api, loaded = self._save_and_load( + tmp_path, group='B01001', variables=variables, sumlevel='county', + ) + assert loaded.name == api.name + assert loaded.scope.name == 'franklin' + assert loaded.sumlevel.name == 'county' + assert loaded.group.code == 'B01001' + assert set(loaded.variables) == set(variables)