From ed47e91a35f6b468f48544e5c0e788bb63a2252f Mon Sep 17 00:00:00 2001 From: Adam Porr Date: Fri, 11 Sep 2026 12:03:39 -0400 Subject: [PATCH 1/5] Add sumlevel for library districts (M31) --- morpc/morpc.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/morpc/morpc.py b/morpc/morpc.py index c244006..9b007ec 100644 --- a/morpc/morpc.py +++ b/morpc/morpc.py @@ -871,6 +871,15 @@ "idField":"REGIONSWACOID", "nameField":"REGIONSWACO", "censusQueryName": None + }, + 'M31': { + "singular":"library district", + "plural":"library districts", + "hierarchy_string":"LIBRARYD", + "authority":"morpc", + "idField":"LIBRARYDID", + "nameField":"LIBRARYD", + "censusQueryName": None }, } From b5396a09a2a1e499e1aa1c9884e782821769d758 Mon Sep 17 00:00:00 2001 From: Adam Porr Date: Thu, 17 Sep 2026 15:31:30 -0400 Subject: [PATCH 2/5] Add morpc.insights module and standalone script to assemble tileset package --- morpc/__init__.py | 1 + morpc/insights/__init__.py | 1 + morpc/insights/insights.py | 578 ++++++++++++++++++ .../insights_assemble_tileset_package.py | 89 +++ 4 files changed, 669 insertions(+) create mode 100644 morpc/insights/__init__.py create mode 100644 morpc/insights/insights.py create mode 100644 morpc/insights/insights_assemble_tileset_package.py diff --git a/morpc/__init__.py b/morpc/__init__.py index f00bb09..8e5d95f 100644 --- a/morpc/__init__.py +++ b/morpc/__init__.py @@ -13,3 +13,4 @@ import morpc.rest_api import morpc.osm import morpc.utils +import morpc.insights diff --git a/morpc/insights/__init__.py b/morpc/insights/__init__.py new file mode 100644 index 0000000..e9113f7 --- /dev/null +++ b/morpc/insights/__init__.py @@ -0,0 +1 @@ +from .insights import * diff --git a/morpc/insights/insights.py b/morpc/insights/insights.py new file mode 100644 index 0000000..9fe7690 --- /dev/null +++ b/morpc/insights/insights.py @@ -0,0 +1,578 @@ +""" +Functions for assembling and validating tileset packages for the MORPC +Insights platform +Reference: https://github.com/morpc/morpc-insights +""" + +import logging +import frictionless + +logger = logging.getLogger(__name__) + +class DataPackage(frictionless.Package): + _requiredProps = ["resources","name","title","description","version","contributors","_vintage","_updateInterval","_techDetailsUrl"] + + def from_descriptor(self, descriptor): + import os + self.basepath = os.path.normpath(os.path.dirname(descriptor)) + package = super().from_descriptor(descriptor) + package.__class__ = DataPackage + return package + + def get_property(self, prop): + if(prop[0] == "_"): + return self.custom[prop] + else: + return getattr(self, prop) + + def set_property(self, prop, value): + if(prop[0] == "_"): + self.custom[prop] = value + else: + setattr(self, prop, value) + + def is_complete(self): + # Verify that required properties have been defined + definedProps = self.list_defined() + for key in self.custom.keys(): + definedProps.append(key) + missingProps = set(self._requiredProps).difference(set(definedProps)) + if(len(missingProps) > 0): + logger.error(f"The following required properties are not defined: {missingProps}") + return False + else: + return True + + complete = True + if(not "table" in self.resources): + logger.error(f"Resources must include one resource named 'table' which provides the metadata for the long-form data table") + complete = False + if(not "process" in self.resource_names): + logger.error(f"Resources must include one resource named 'process' which provides the metadata for a document describing the process by which the data was produced.") + complete = False + + return complete + + +class PresentationPackage(frictionless.Package): + _requiredProps = ["resources","name","title","description","version","contributors","_visualizationSpec","_visualizationSpecOverride","_moreContextUrl"] + + def from_descriptor(self, descriptor): + import os + self.basepath = os.path.normpath(os.path.dirname(descriptor)) + package = super().from_descriptor(descriptor) + package.__class__ = PresentationPackage + return package + + def get_property(self, prop): + if(prop[0] == "_"): + return self.custom[prop] + else: + return getattr(self, prop) + + def set_property(self, prop, value): + if(prop[0] == "_"): + self.custom[prop] = value + else: + setattr(self, prop, value) + + def is_complete(self): + # Verify that required properties have been defined + definedProps = self.list_defined() + for key in self.custom.keys(): + definedProps.append(key) + missingProps = set(self._requiredProps).difference(set(definedProps)) + if(len(missingProps) > 0): + logger.error(f"The following required properties are not defined: {missingProps}") + return False + else: + return True + + complete = True + if(not "commentary" in self.resource_names): + logger.error(f"Resources must include one resource named 'commentary' which provides a descriptor for the commentary table") + complete = False + + return complete + +class TilesetPackage(frictionless.Package): + _nativeProps = ["name","title","description","version"] + _requiredProps = ["resources","name","title","description","version","contributors","_vintage", + "_updateInterval","_techDetailsUrl","_visualizationSpec","_visualizationSpecOverride", + "_moreContextUrl"] + + def from_descriptor(self, descriptor): + import os + self.basepath = os.path.normpath(os.path.dirname(descriptor)) + package = super().from_descriptor(descriptor) + package.__class__ = PresentationPackage + return package + + def props_from_packages(self, dataPackage, presentationPackage): + self.resources = [] + self.dataPackageBasepath = dataPackage.basepath + self.presentationPackageBasepath = presentationPackage.basepath + for prop in dataPackage._requiredProps: + if((prop in self._nativeProps) or (prop == "resources")): + continue + else: + self.set_property(prop, dataPackage.get_property(prop)) + for prop in presentationPackage._requiredProps: + if((prop in self._nativeProps) or (prop == "resources")): + continue + else: + self.set_property(prop, presentationPackage.get_property(prop)) + + def get_property(self, prop): + if(prop[0] == "_"): + return self.custom[prop] + else: + return getattr(self, prop) + + def set_property(self, prop, value): + if(prop[0] == "_"): + self.custom[prop] = value + else: + setattr(self, prop, value) + + def is_complete(self): + # Verify that required properties have been defined + definedProps = self.list_defined() + for key in self.custom.keys(): + definedProps.append(key) + missingProps = set(self._requiredProps).difference(set(definedProps)) + if(len(missingProps) > 0): + logger.error(f"The following required properties are not defined: {missingProps}") + return False + + complete = True + if(not "catalog" in self.resource_names): + logger.error(f"Resources must include one resource named 'catalog' which provides the metadata for an Excel document which includes the details required for this tileset for inclusion in the Insights platform catalog") + complete = False + if(not "table" in self.resource_names): + logger.error(f"Resources must include one resource named 'table' which provides the metadata for the long-form data table") + complete = False + if(not "process" in self.resource_names): + logger.error(f"Resources must include one resource named 'process' which provides the metadata for a document describing the process by which the data was produced.") + complete = False + if(not "readme" in self.resource_names): + logger.error(f"Resources must include one resource named 'readme' which provides the metadata for a document (ideally named README.md and in Markdown format) which provides human-readable notes and metadata for this tileset.") + complete = False + if(not "maintainer" in self.resource_names): + logger.error(f"Resources must include one resource named 'maintainer' which provides the metadata for a document (ideally named MAINTAINER.md and in Markdown format) which describes the process to update the tileset.") + complete = False + + return complete + +def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", dataPackagePath=None, presentationPackagePath=None, tilesetVersion=None, tilesetDescription=None, thumbnailUrlBase=None): + """ + Given an Insights Data Package and an Insights Presentation Package, create an Insights Tileset Package including the associated + Frictionless package file and all of metadata and artifacts provided in the two input packages. + + Parameters + ---------- + tilesetSlug : str + A short string that uniquely identifies the tileset. Typically the GitHub repository slug, e.g. "insights-pop". + tilesetTitle : str + The title for the tileset package. This will be combined with TITLE_PREFIX (see below) and the result will be + used for the Frictionless "title" property + tilesetBasepath : str + A string representing the path to the root directory for the tileset contents. This directory will be created if + it doesn't already exist. If no path is provided, the current working directory will be used. + dataPackagePath : str + A string representing the path to the Frictionless YAML file that defines an Insights Data Package. This should + be present in the root folder of the Data Package and should have the name "data.package.yaml". If no path is provided, + the script will look for "data.package.yaml" in in the directory specified for tilesetBasepath. + presentationPackagePath : str + A string representing the path to the Frictionless YAML file that defines an Insights Presentation Package. This should + be present in the root folder of the Presentation Package and should have the name "presentation.package.yaml". If no + path is provided, the script will look for "presentation.package.yaml" in in the directory specified for tilesetBasepath. + tilesetVersion : str + A string that uniquely identifies this version of the tileset. Typically using the format YYYY.mm.dd. If tilesetVersion + is not specified by the user, it will be constructed using the current date. + tilesetDescription : str + A brief description of the subject matter covered by the tileset. Will be used for the Frictionless "description" + property. If tilesetDescription is not specified by the user, the description from the Presentation Package will be + used. + thumbnailUrlBase : str + The base URL by which the thumbnail images will be accessed. Each thumbnail should be accessible by appending the thumbnail + filename to the base URL. Typically the URL would point to a GitHub repository. If thumbnailUrlBase is not specified, the script + will assume that GitHub is being used and will construct a URL using the tilesetSlug and the figures directory (see below) + + Returns + ------- + tp : morpc.insights.TilesetPackage + An object representing an Insights Tileset Package. Includes metadata from the referenced Data Package and + Presentation Package + """ + # ## Setup + # ### Import required packages + import morpc + import frictionless + import datetime + import xlsxwriter + import pandas as pd + import os + import posixpath + import shutil + import logging + + # ### Static parameters + # Prefix that will be applied to the tileset title + TITLE_PREFIX = "MORPC Insights" + # Subdirectory of tileset directory where output data will be stored + OUTPUT_DATA_DIR = "output_data" + # Subdirectory of tileset directory where input data will be stored + INPUT_DATA_DIR = "input_data" + # Subdirectory of tileset directory where figures will be stored + FIGURES_DIR = "figures" + # Subdirectory of tileset directory where non-required will be stored + MISC_DIR = "misc" + # Default file name to use for data package if no path is provided + DEFAULT_DATAPACKAGE_FILENAME = "data.package.yaml" + # Default file name to use for presentation package if no path is provided + DEFAULT_PRESENTATIONPACKAGE_FILENAME = "presentation.package.yaml" + # Filename for the tileset package when written to disk + TILESET_PACKAGE_FILENAME = "tileset.package.yaml" + + # ### Start logging + logger = logging.getLogger(__name__) + + # ### Interpret arguments + if(dataPackagePath is None): + dataPackagePath = os.path.normpath(os.path.join("./", DEFAULT_DATAPACKAGE_FILENAME)) + logger.info(f"Data Package path was not specified. Using default path {dataPackagePath}") + if(presentationPackagePath is None): + presentationPackagePath = os.path.normpath(os.path.join("./", DEFAULT_PRESENTATIONPACKAGE_FILENAME)) + logger.info(f"Presentation Package path was not specified. Using default path {presentationPackagePath}") + if(tilesetVersion is None): + tilesetVersion = datetime.datetime.now().strftime("%Y.%m.%d") + logger.info(f"Tileset version was not specified. Using version derived from today's date {tilesetVersion}") + if(thumbnailUrlBase is None): + thumbnailUrlBase = f"https://raw.githubusercontent.com/morpc-insights/{tilesetSlug}/refs/heads/main/{FIGURES_DIR}/" + logger.info(f"Thumbnail base URL was not specified. Using URL constructed from the tileset slug and the figures directory name: {thumbnailUrlBase}") + + # ## Prepare inputs + # ### Read data package configuration + dp = morpc.insights.DataPackage().from_descriptor(dataPackagePath) + results = dp.validate() + if(not results.valid): + logger.error("Input Data Package is not a valid Frictionless Package. Details follow.") + logger.error(results) + raise RuntimeError + if(not dp.is_complete()): + logger.error("Input Data Package is missing one or more required elements.") + raise RuntimeError + else: + logger.info(f"Reading Data Package at {dataPackagePath}") + + # ### Read presentation package configuration + pp = morpc.insights.PresentationPackage().from_descriptor(presentationPackagePath) + results = pp.validate() + if(not results.valid): + logger.error("Input Presentation Package is not a valid Frictionless Package. Details follow.") + logger.error(results) + raise RuntimeError + if(not pp.is_complete()): + logger.error("Input Presentation Package is missing one or more required elements.") + raise RuntimeError + else: + logger.info(f"Reading Presentation Package at {presentationPackagePath}") + + if(tilesetDescription is None): + tilesetDescription = pp.description + logger.info(f"Tileset description was not specified. Using description provided in Presentation Package") + + # ## Create tileset package + # ### Create tileset directory structure + logger.info("Creating tileset directory structure (as needed)") + tilesetBasepath = os.path.normpath(tilesetBasepath) + tilesetOutputDataPath = os.path.join(tilesetBasepath, OUTPUT_DATA_DIR) + tilesetInputDataPath = os.path.join(tilesetBasepath, INPUT_DATA_DIR) + tilesetFiguresPath = os.path.join(tilesetBasepath, FIGURES_DIR) + tilesetMiscPath = os.path.join(tilesetBasepath, MISC_DIR) + if not os.path.exists(tilesetBasepath): + os.makedirs(tilesetBasepath) + if not os.path.exists(tilesetOutputDataPath): + os.makedirs(tilesetOutputDataPath) + if not os.path.exists(tilesetInputDataPath): + os.makedirs(tilesetInputDataPath) + if not os.path.exists(tilesetFiguresPath): + os.makedirs(tilesetFiguresPath) + if not os.path.exists(tilesetMiscPath): + os.makedirs(tilesetMiscPath) + + # ### Define tileset package properties + logger.info("Defining tileset package properties") + tp = morpc.insights.TilesetPackage() + tp.props_from_packages(dp, pp) + tp.basepath = tilesetBasepath + tp.set_property("name", tilesetSlug) + tp.set_property("title", tilesetTitle) + tp.set_property("description", tilesetDescription) + tp.set_property("version", tilesetVersion) + tp.dataPackageBasepath = posixpath.dirname(dataPackagePath) + tp.presentationPackageBasepath = posixpath.dirname(presentationPackagePath) + tp.resources = [] + + # ### Add resources to tileset package + # #### README + logger.info("Adding README.md resource to tileset package.") + readmeResource = frictionless.Resource() + readmeResource.name = "readme" + readmeResource.title= f"{TITLE_PREFIX} | {tilesetTitle} | README" + readmeResource.description = "This document provides an overview and metadata about the insights-pop (Historic and Forecasted Population by Year) tileset in a human-readable form." + readmeResource.path = "README.md" + tp.add_resource(readmeResource); + + # #### Maintainer guide + logger.info("Adding Maintainer Guide resource to tileset package.") + maintainerResource = frictionless.Resource() + maintainerResource.name = "maintainer" + maintainerResource.title= f"{TITLE_PREFIX} | {tilesetTitle} | Maintainer Guide" + maintainerResource.description = f"This document describes how to update the {tilesetSlug} ({tilesetTitle}) tileset." + maintainerResource.path = "MAINTAINER.md" + tp.add_resource(maintainerResource); + + # #### Resources from data package + # Add resources to tileset package. + logger.info("Adding resources defined in data package.") + for resource in dp.resources: + if(resource.name in tp.resource_names): + logger.error(f"Resource {resource.name} already defined in tileset package") + raise RuntimeError + else: + logger.info(f"--> Resource: {resource.name}") + if((resource.name == "table") or (resource.name.find("output") != -1)): + # Put the data table (and schema and resource file if available) in the data directory. If there are other outputs, do the same for those. + destinationDir = posixpath.normpath(OUTPUT_DATA_DIR) + elif(resource.name.find("process") != -1): + # Put any process definitions (i.e. any resources whose names include "process") in the root directory + destinationDir = "" + elif(resource.name.find("input") != -1): + # Put any input data (and schema and resource file if available) in the input data directory + destinationDir = posixpath.normpath(INPUT_DATA_DIR) + else: + # Put all other resources in the misc direcotry + destinationDir = posixpath.normpath(MISC_DIR) + resource.path = posixpath.join(destinationDir, posixpath.basename(resource.path)) + tp.add_resource(resource) + + # Copy files specified as resources from the data package file structure to the tileset package file structure. For data files, attempt + # to copy the schema and resource files too, if available. + for resourceName in dp.resource_names: + logger.info(f"Copying files associated with Data Package resource '{resourceName}'") + resource = dp.get_resource(resourceName) + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, resource.path)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, resource.path)) + + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + + # For the long form table (required) the and any resources designated as inputs or outputs (optional), we also need to copy + # the schema and standalone resource file. The schema is required for the long form table, but optional for others. + if((resourceName == "table") or (resource.name.find("input") != -1) or (resource.name.find("output") != -1)): + # In the next line, I would think you could access the schema filename as resource.schema, + # however frictionless seems to dereference the schema automatically and it returns a + # schema object. It is possible to get the schema filename through metadata_export, but + # maybe there is a better way? + if("schema" in resource.metadata_export()): + schemaPath = resource.metadata_export()["schema"] + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, schemaPath)) + if(os.path.exists(sourcePath)): + logger.info(f"--> Detected schema associated with resource {resourceName}. Copying schema from data package to tileset package") + destinationPath = os.path.normpath(os.path.join(tp.basepath, schemaPath)) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + else: + logger.error(f"--> Schema defined for resource '{resource}' but specified file does not exist ({sourcePath})") + raise RuntimeError + else: + if(resourceName == "table"): + logger.error(f"--> Schema is required for resource 'table' but schema path was not specified.") + raise RuntimeError + else: + logger.warning(f"--> Schema path not specified for resource '{resourceName}'. Schema will not be included in Tileset Package.") + + # There may not be a standalone resource file because the resource information is required to be captured in the package file. + # Guess at the name of a standalone resource. If it exists, we'll grab that too. Our guess assumes that the resource file is located + # in the same directory as the data file and is named similarly but with extension ".resource.yaml" + extension = os.path.splitext(resource.path)[1] + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, resource.path.replace(extension,".resource.yaml"))) + if(os.path.exists(sourcePath)): + logger.info("--> Detected standalone resource file. Copying resource file from data package to tileset package") + destinationPath = os.path.normpath(os.path.join(tp.basepath, resource.path.replace(extension,".resource.yaml"))) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + + # #### Resources from presentation package + # Add resources to tileset package. + logger.info("Adding resources defined in presentation package.") + for resource in pp.resources: + if(resource.name in tp.resource_names): + logger.error(f"Resource {resource.name} already defined in tileset package") + raise RuntimeError + else: + logger.info(f"--> Resource: {resource.name}") + if(resource.name == "catalog"): + # Put the catalog in the root directory + destinationDir = "" + else: + # Put all other resources in the misc direcotry + destinationDir = posixpath.normpath(os.path.join(tp.basepath, MISC_DIR)) + resource.path = posixpath.join(destinationDir, posixpath.basename(resource.path)) + tp.add_resource(resource) + + # Copy files specified as resources from the presentation package file structure to the tileset package file structure. + resource_names = pp.resource_names + # We will not copy the catalog. Rather, we'll read it, make some adjustments, and write the adjusted version + resource_names.remove("catalog") + for resourceName in pp.resource_names: + logger.info(f"Copying files associated with Presentation Package resource '{resourceName}'") + resource = pp.get_resource(resourceName) + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, resource.path)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, resource.path)) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + + # ## Update catalog + logger.info("Loading catalog provided with presentation package") + catalogResource = pp.get_resource("catalog") + catalog = pd.read_excel(catalogResource.path) + + # Ensure that all values expected in the Presentation Package have been populated. + missingFlag = False + for column in ["GeographyType","GeographyName","Headline","Commentary","ThumbnailURL","DataProductURL"]: + if(not catalog.loc[catalog[column].isna()].empty): + missingFlag = True + if(column == "Headline"): + additionalInstructions = "The headlines for community-level geographies may be identical." + elif(column == "Commentary"): + additionalInstructions = "The commentary for community-level geographies may be identical." + elif(column == "ThumbnailURL"): + additionalInstructions = "Please enter URLs or paths for all geographies. If a local path is provided it will be converted to a URL using the thumbnailUrlBase value." + elif(column == "DataProductURL"): + additionalInstructions = "Please enter a data product URLs for all geographies. Each geography may have a unique URL, or any set of geographies may share a common URL." + logger.error(f"Blank values detected in column {column}. Please enter a value for for all geographies. {additionalInstructions}") + if(missingFlag == True): + raise RuntimeError + else: + logger.info("No missing values detected.") + + # Update thumbnail URLs and copy provided images from the presentation package to the tileset package if necessary + visualizationSpec = pp.get_property("_visualizationSpec") + if(visualizationSpec == "useProvidedCopy"): + logger.info("Creating copies of the thumbnails in the tileset package.") + filenames = catalog["ThumbnailURL"].apply(lambda x:os.path.basename(x)) + catalog["ThumbnailURL"] = thumbnailUrlBase + filenames + showNotification = True + for file in filenames: + sourcePath = os.path.normpath(os.path.join(tp.presentationPackageBasepath, FIGURES_DIR, file)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, FIGURES_DIR, file)) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + if(showNotification == True): + logger.info(f"Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file. Notifications will be suppressed for remaining files.") + showNotification = False + else: + shutil.copyfile(sourcePath, destinationPath) + elif(visualizationSpec == "useProvidedInPlace"): + # In this case, we'll use the provided URL as-is. Make no changes. + logger.info("Using provided thumbnail URLs as-is. No changes required") + else: + # Eventually we will support construction of figures from a visualization specification. As of September 2026, + # this is not implemented. Return an error. + logger.error("Only visualizationSpec values useProvidedCopy and useProvidedInPlace are currently supported.") + raise RuntimeError + + # Ensure TileID, TilesetID, Category, and ShareURL are blank. These are managed by a downstream workflow. + catalog["TileID"] = None + catalog["TilesetID"] = None + catalog["Category"] = None + catalog["ShareURL"] = None + + # Populate Contributor, Vintage, and UpdateInterval from metadata originally from the Data Package + catalog["Contributor"] = tp.contributors[0]["title"] + catalog["Vintage"] = tp.get_property("_vintage") + catalog["UpdateInterval"] = tp.get_property("_updateInterval") + + # Write the updated catalog to disk in the Tileset Package file structure + logger.info("Writing updated catalog to disk") + sourcePath = os.path.normpath(os.path.join(tp.presentationPackageBasepath, catalogResource.path)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, os.path.basename(catalogResource.path))) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Existing catalog will be updated in-place.") + else: + logger.info(f"Destination path ({os.path.abspath(destinationPath)}) is different than source path ({os.path.abspath(sourcePath)}). Catalog provided with Presentation Package will be left unaltered.") + catalog.to_excel(destinationPath, index=False) + + # ### Validate tileset package + logger.info("Checking completeness of Tileset Package object") + if(not tp.is_complete()): + logger.error("--> Tileset package object is missing one or more required elements") + raise RuntimeError + else: + logger.info("--> Tileset package object is complete") + + # ## Create README.md file + logger.info("Creating README.md file") + + README_TEMPLATE = f''' + # {tp.title} + + ## Version + + Current version: {tp.version} + + ## Contributors + + {"\n".join([f"{x['title']}, {x['organization']}, {x['email']}" for x in tp.contributors])} + + ## Introduction + + {tp.description} + + ## Data + + Data file: {tp.get_resource("table").path} + + Schema: {schemaPath} + + ## Processes + + Process documentation: {tp.get_resource("process").path} + ''' + + with open(os.path.join(tilesetBasepath, readmeResource.path), "w") as f: + f.write(README_TEMPLATE) + + # ## Write tileset package to disk and validate it + logger.info(f"Writing tileset package descriptor to disk") + destinationPath = os.path.normpath(os.path.join(tilesetBasepath, TILESET_PACKAGE_FILENAME)) + tp.to_yaml(destinationPath); + results = morpc.frictionless.validate(destinationPath) + if(not results.valid): + logger.error("Tileset Package is not a valid Frictionless Package. Details follow.") + logger.error(results) + + # ## Clean up + # Delete the misc directory if it is empty. Not all tilesets include misc content. Same with input data directory. + if(len(os.listdir(tilesetMiscPath)) == 0): + os.rmdir(tilesetMiscPath) + if(len(os.listdir(tilesetInputDataPath)) == 0): + os.rmdir(tilesetInputDataPath) + + return tp + + \ No newline at end of file diff --git a/morpc/insights/insights_assemble_tileset_package.py b/morpc/insights/insights_assemble_tileset_package.py new file mode 100644 index 0000000..92c9614 --- /dev/null +++ b/morpc/insights/insights_assemble_tileset_package.py @@ -0,0 +1,89 @@ +# insights_assemble_tileset_package.py +# +# This script is a wrapper for morpc.insights.assemble_tileset_package(). Given an Insights Data Package +# and an Insights Presentation Package, create an Insights Tileset Package including the associated +# Frictionless package file and all of metadata and artifacts provided in the two input packages. + +import morpc +import logging +import argparse +import sys +import os + +LOGLEVEL = "info" + +LOGFORMAT = '%(asctime)s | %(levelname)s | %(name)s.%(funcName)s: %(message)s' + +LEVEL_MAP = { + "debug": 10, + "info": 20, + "warning": 30, + "error": 40, + "critical": 50 +} + +logging.basicConfig( + level=LEVEL_MAP[LOGLEVEL], + force=True, + format=LOGFORMAT, + handlers=[ + logging.StreamHandler(sys.stdout) + ] + ) +logging.getLogger(__name__).setLevel(LEVEL_MAP[LOGLEVEL]) + +epilogString = ''' +# Examples + +## Create tileset package in-situ (most common, least complex) + +Typically, you\'ll already have a GitHub repository which contains the Data Package elements and Presentation Package +elements and you just want to create the tileset Frictionless metadata (tileset.package.yaml) and update metadata in +the catalog. In that case, you can rely on the defaults for most arguments. + +NOTE: Your catalog.xlsx will be overwritten in this case and some values will be altered. + +python insights_assemble_tileset_package.py "insights-pop" "Historic and Forecasted Population by Year" "C:\\Users\\yourname\\github\\insights-pop" + +## Create separate tileset package in new folder using all optional arguments (for reference only, real use cases are uncommon) + +python insights_assemble_tileset_package.py "insights-pop" "Historic and Forecasted Population by Year" "C:\\Users\\yourname\\github\\insights-pop" --dataPackagePath "C:\\some_path_to\\data.package.yaml" --presentationPackagePath "C:\\another_path_to\\presentation.package.yaml" --tilesetVersion "2026.09.17" --tilesetDescription "I didn\'t like the description in the Presentation Package so I wrote this one instead" --thumbnailUrlBase "https://www.myserver.com/somedir/" +''' + +parser = argparse.ArgumentParser( + description = "This script is a wrapper for morpc.insights.assemble_tileset_package(). Given an Insights Data Package and an Insights Presentation Package, create an Insights Tileset Package including the associated Frictionless package file and all of metadata and artifacts provided in the two input packages.", + epilog = epilogString.replace("\\\\","\\"), + formatter_class=argparse.RawDescriptionHelpFormatter +) +parser.add_argument("tilesetSlug", + help = "A short string that uniquely identifies the tileset. Typically the GitHub repository slug, e.g. 'insights-pop'." +) +parser.add_argument("tilesetTitle", + help = "The title for the tileset package. This will be combined with TITLE_PREFIX (see morpc.insights code) and the result will be used for the Frictionless 'title' property") +parser.add_argument("tilesetBasepath", + help = "A string representing the path to the root directory for the tileset contents. This directory will be created if it doesn't already exist. If no path is provided, the current working directory will be used." +) +parser.add_argument("--dataPackagePath", + help = "A string representing the path to the Frictionless YAML file that defines an Insights Data Package. This should be present in the root folder of the Data Package and should have the name 'data.package.yaml'. If no path is provided, the script will look for 'data.package.yaml' in in the directory specified for tilesetBasepath." +) +parser.add_argument("--presentationPackagePath", + help = "A string representing the path to the Frictionless YAML file that defines an Insights Presentation Package. This should be present in the root folder of the Presentation Package and should have the name 'presentation.package.yaml'. If no path is provided, the script will look for 'presentation.package.yaml' in in the directory specified for tilesetBasepath." +) +parser.add_argument("--tilesetVersion", + help = "A string that uniquely identifies this version of the tileset. Typically using the format YYYY.mm.dd. If tilesetVersion is not specified by the user, it will be constructed using the current date." +) +parser.add_argument("--tilesetDescription", + help = "A brief description of the subject matter covered by the tileset. Will be used for the Frictionless 'description' property. If tilesetDescription is not specified by the user, the description from the Presentation Package will be used." +) +parser.add_argument("--thumbnailUrlBase", + help = "The base URL by which the thumbnail images will be accessed. Each thumbnail should be accessible by appending the thumbnail filename to the base URL. Typically the URL would point to a GitHub repository. If thumbnailUrlBase is not specified, the script will assume that GitHub is being used and will construct a URL using the tilesetSlug and the figures directory (see morpc.insights code)" +) +args = parser.parse_args() + +morpc.insights.assemble_tileset_package(args.tilesetSlug, args.tilesetTitle, os.path.normpath(args.tilesetBasepath), + dataPackagePath = args.dataPackagePath, + presentationPackagePath = args.presentationPackagePath, + tilesetVersion = args.tilesetVersion, + tilesetDescription = args.tilesetDescription, + thumbnailUrlBase = args.thumbnailUrlBase +) From 0851893b87b5dd99f5e53899ee725f49587f8c3c Mon Sep 17 00:00:00 2001 From: Adam Porr Date: Thu, 24 Sep 2026 12:08:18 -0400 Subject: [PATCH 3/5] In morpc.data_chart_to_excel add custom handling of null values and set defaults for scatter plots --- morpc/morpc.py | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/morpc/morpc.py b/morpc/morpc.py index 9b007ec..b8b0608 100644 --- a/morpc/morpc.py +++ b/morpc/morpc.py @@ -2278,7 +2278,7 @@ def recursiveUpdate(original, updates): original[key] = value return original -def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dataOptions=None, chartOptions=None): +def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dataOptions=None, chartOptions=None, na_rep=None): # TODO: simplify docstring """ Create an Excel worksheet consisting of the contents of a pandas dataframe (as a formatted table) @@ -2423,6 +2423,9 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat "y2AxisOptions": dict Options to control the appearance of the secondary y axis. Will be used directly by chart.set_y2_axis(). No defaults are applied. All required values must by specified. See https://xlsxwriter.readthedocs.io/chart.html#chart-set-y2-axis + na_rep: str + Value to use represent null values in the data frame. Default is ''. Use '=NA()' to use Excel-compatible + null values. This is necessary when including null values in ranges to be shown in a line chart. Returns ------- @@ -2525,11 +2528,13 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat }, "smooth": False }) + seriesOptionsDefault["scatter"] = json.loads(json.dumps(seriesOptionsDefault["line"])) subtypesDefaults = { "bar": None, "column": None, - "line": None + "line": None, + "scatter": None } myDataOptions = { @@ -2583,7 +2588,7 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat workbook = writer.book - df.to_excel(writer, sheet_name=sheet_name, index=myDataOptions["index"]) + df.to_excel(writer, sheet_name=sheet_name, index=myDataOptions["index"], na_rep=na_rep) worksheet = writer.sheets[sheet_name] @@ -2677,7 +2682,6 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat color = myChartOptions["colors"][(i-1) % len(myChartOptions["colors"])] elif(type(myChartOptions["colors"]) == dict): color = myChartOptions["colors"].get(colname, styleDefaults["seriesColor"]) # Revert to default if color is not specified for column - json.dumps(mySeriesOptions, indent=4) # Else if we have more than one series, cycle through the default set of colors elif(nColumns > 1): color = colorsDefault[(i-1) % len(colorsDefault)] @@ -2741,7 +2745,7 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat # the y-axis title was not provided in the dict, revert to the default. if(type(myChartOptions["titles"]) == dict): myYAxisOptions["name"] = myChartOptions["titles"].get("yTitle", yAxisOptionsDefaults["name"]) - + if(colname in y2AxisColumns): mySeriesOptions["y2_axis"] = True chart.add_series(mySeriesOptions) From 3ac610f6d9e017a200c397c05070cf5ac4296979 Mon Sep 17 00:00:00 2001 From: Adam Porr Date: Wed, 30 Sep 2026 12:30:40 -0400 Subject: [PATCH 4/5] Update morpc.insights to reflect changes to tileset specification --- morpc/insights/insights.py | 169 +++++++++++++++++++++++-------------- 1 file changed, 106 insertions(+), 63 deletions(-) diff --git a/morpc/insights/insights.py b/morpc/insights/insights.py index 9fe7690..ea5e855 100644 --- a/morpc/insights/insights.py +++ b/morpc/insights/insights.py @@ -44,8 +44,8 @@ def is_complete(self): return True complete = True - if(not "table" in self.resources): - logger.error(f"Resources must include one resource named 'table' which provides the metadata for the long-form data table") + if(not "output" in self.resources): + logger.error(f"Resources must include one resource named 'output' which provides the metadata for the long-form data table") complete = False if(not "process" in self.resource_names): logger.error(f"Resources must include one resource named 'process' which provides the metadata for a document describing the process by which the data was produced.") @@ -89,8 +89,8 @@ def is_complete(self): return True complete = True - if(not "commentary" in self.resource_names): - logger.error(f"Resources must include one resource named 'commentary' which provides a descriptor for the commentary table") + if(not "catalog" in self.resource_names): + logger.error(f"Resources must include one resource named 'catalog' which provides a descriptor for the catalog table") complete = False return complete @@ -149,7 +149,7 @@ def is_complete(self): if(not "catalog" in self.resource_names): logger.error(f"Resources must include one resource named 'catalog' which provides the metadata for an Excel document which includes the details required for this tileset for inclusion in the Insights platform catalog") complete = False - if(not "table" in self.resource_names): + if(not "output" in self.resource_names): logger.error(f"Resources must include one resource named 'table' which provides the metadata for the long-form data table") complete = False if(not "process" in self.resource_names): @@ -164,7 +164,7 @@ def is_complete(self): return complete -def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", dataPackagePath=None, presentationPackagePath=None, tilesetVersion=None, tilesetDescription=None, thumbnailUrlBase=None): +def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", dataPackagePath=None, presentationPackagePath=None, tilesetVersion=None, tilesetDescription=None, thumbnailUrlBase=None, produceReadme=False): """ Given an Insights Data Package and an Insights Presentation Package, create an Insights Tileset Package including the associated Frictionless package file and all of metadata and artifacts provided in the two input packages. @@ -198,6 +198,9 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da The base URL by which the thumbnail images will be accessed. Each thumbnail should be accessible by appending the thumbnail filename to the base URL. Typically the URL would point to a GitHub repository. If thumbnailUrlBase is not specified, the script will assume that GitHub is being used and will construct a URL using the tilesetSlug and the figures directory (see below) + produceReadme : bool + If set to True, the function will automatically produce a basic README.md file using the available metadata. By + default, no README.md file will be produced. Returns ------- @@ -214,6 +217,7 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da import pandas as pd import os import posixpath + import pathlib import shutil import logging @@ -311,8 +315,8 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da tp.set_property("title", tilesetTitle) tp.set_property("description", tilesetDescription) tp.set_property("version", tilesetVersion) - tp.dataPackageBasepath = posixpath.dirname(dataPackagePath) - tp.presentationPackageBasepath = posixpath.dirname(presentationPackagePath) + tp.dataPackageBasepath = posixpath.dirname(pathlib.PurePath(dataPackagePath).as_posix()) + tp.presentationPackageBasepath = posixpath.dirname(pathlib.PurePath(presentationPackagePath).as_posix()) tp.resources = [] # ### Add resources to tileset package @@ -321,7 +325,7 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da readmeResource = frictionless.Resource() readmeResource.name = "readme" readmeResource.title= f"{TITLE_PREFIX} | {tilesetTitle} | README" - readmeResource.description = "This document provides an overview and metadata about the insights-pop (Historic and Forecasted Population by Year) tileset in a human-readable form." + readmeResource.description = f"This document provides an overview of the {tilesetSlug} ({tilesetTitle}) tileset and its components." readmeResource.path = "README.md" tp.add_resource(readmeResource); @@ -335,7 +339,7 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da tp.add_resource(maintainerResource); # #### Resources from data package - # Add resources to tileset package. + # Add resources to tileset package descriptor. logger.info("Adding resources defined in data package.") for resource in dp.resources: if(resource.name in tp.resource_names): @@ -343,20 +347,26 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da raise RuntimeError else: logger.info(f"--> Resource: {resource.name}") - if((resource.name == "table") or (resource.name.find("output") != -1)): - # Put the data table (and schema and resource file if available) in the data directory. If there are other outputs, do the same for those. - destinationDir = posixpath.normpath(OUTPUT_DATA_DIR) + if(resource.name.find("output") != -1): + # Put the primary output table (and schema and resource file if available) in the output data directory. If there are other outputs, do the same for those. + destinationDir = OUTPUT_DATA_DIR elif(resource.name.find("process") != -1): # Put any process definitions (i.e. any resources whose names include "process") in the root directory destinationDir = "" elif(resource.name.find("input") != -1): # Put any input data (and schema and resource file if available) in the input data directory - destinationDir = posixpath.normpath(INPUT_DATA_DIR) + destinationDir = INPUT_DATA_DIR else: - # Put all other resources in the misc direcotry - destinationDir = posixpath.normpath(MISC_DIR) - resource.path = posixpath.join(destinationDir, posixpath.basename(resource.path)) - tp.add_resource(resource) + # Put all other resources in the misc directory + destinationDir = MISC_DIR + newResource = resource.to_copy() + if("_cache" in resource.custom): + newResource.custom["_cache"] = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.custom["_cache"]).as_posix())) + else: + newResource.path = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.path).as_posix())) + if("schema" in resource.metadata_export()): + newResource.schema = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.metadata_export()["schema"]).as_posix())) + tp.add_resource(newResource) # Copy files specified as resources from the data package file structure to the tileset package file structure. For data files, attempt # to copy the schema and resource files too, if available. @@ -423,19 +433,26 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da raise RuntimeError else: logger.info(f"--> Resource: {resource.name}") - if(resource.name == "catalog"): - # Put the catalog in the root directory + newResource = resource.to_copy() + if(resource.name == "commentary"): + # We'll use the commentary spreadsheet as the basis for the tileset catalog, which + # goes in the root directory destinationDir = "" + newResource.path = posixpath.join(destinationDir, "catalog.xlsx") + newResource.schema = posixpath.join(destinationDir, "catalog.schema.yaml") + newResource.name = "catalog" + tp.add_resource(newResource) else: # Put all other resources in the misc direcotry - destinationDir = posixpath.normpath(os.path.join(tp.basepath, MISC_DIR)) - resource.path = posixpath.join(destinationDir, posixpath.basename(resource.path)) - tp.add_resource(resource) + destinationDir = posixpath.join(tp.basepath, MISC_DIR) + newResource.path = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.path).as_posix())) + tp.add_resource(newResource) # Copy files specified as resources from the presentation package file structure to the tileset package file structure. resource_names = pp.resource_names - # We will not copy the catalog. Rather, we'll read it, make some adjustments, and write the adjusted version - resource_names.remove("catalog") + # We will not copy the commentary. Rather, we'll read it, make some adjustments, and write the adjusted version + # as the tileset catalog + resource_names.remove("commentary") for resourceName in pp.resource_names: logger.info(f"Copying files associated with Presentation Package resource '{resourceName}'") resource = pp.get_resource(resourceName) @@ -447,15 +464,15 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") shutil.copyfile(sourcePath, destinationPath) - # ## Update catalog - logger.info("Loading catalog provided with presentation package") - catalogResource = pp.get_resource("catalog") - catalog = pd.read_excel(catalogResource.path) + # ## Create catalog + logger.info("Loading commentary provided with presentation package") + commentaryResource = pp.get_resource("commentary") + commentary = pd.read_excel(commentaryResource.path) # Ensure that all values expected in the Presentation Package have been populated. missingFlag = False - for column in ["GeographyType","GeographyName","Headline","Commentary","ThumbnailURL","DataProductURL"]: - if(not catalog.loc[catalog[column].isna()].empty): + for column in ["GeoType","GeoName","Headline","Commentary","ThumbnailURL","DataProductURL"]: + if(not commentary.loc[commentary[column].isna()].empty): missingFlag = True if(column == "Headline"): additionalInstructions = "The headlines for community-level geographies may be identical." @@ -473,10 +490,10 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da # Update thumbnail URLs and copy provided images from the presentation package to the tileset package if necessary visualizationSpec = pp.get_property("_visualizationSpec") - if(visualizationSpec == "useProvidedCopy"): + if(visualizationSpec == "useProvided"): logger.info("Creating copies of the thumbnails in the tileset package.") - filenames = catalog["ThumbnailURL"].apply(lambda x:os.path.basename(x)) - catalog["ThumbnailURL"] = thumbnailUrlBase + filenames + filenames = commentary["ThumbnailURL"].apply(lambda x:os.path.basename(x)) + commentary["ThumbnailURL"] = thumbnailUrlBase + filenames showNotification = True for file in filenames: sourcePath = os.path.normpath(os.path.join(tp.presentationPackageBasepath, FIGURES_DIR, file)) @@ -487,35 +504,53 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da showNotification = False else: shutil.copyfile(sourcePath, destinationPath) - elif(visualizationSpec == "useProvidedInPlace"): - # In this case, we'll use the provided URL as-is. Make no changes. - logger.info("Using provided thumbnail URLs as-is. No changes required") else: # Eventually we will support construction of figures from a visualization specification. As of September 2026, # this is not implemented. Return an error. logger.error("Only visualizationSpec values useProvidedCopy and useProvidedInPlace are currently supported.") raise RuntimeError - # Ensure TileID, TilesetID, Category, and ShareURL are blank. These are managed by a downstream workflow. + # Create the tileset catalog using the commentary spreadsheet as the base + catalog = commentary.copy() + catalogSchema = tp.get_resource("catalog").schema + + # Create new fields TileID, TilesetID, Category, and ShareURL and leave them blank. These are managed + # by a downstream workflow. catalog["TileID"] = None catalog["TilesetID"] = None catalog["Category"] = None - catalog["ShareURL"] = None + catalog["Priority"] = None - # Populate Contributor, Vintage, and UpdateInterval from metadata originally from the Data Package + # Populate Contributor, Vintage, UpdateInterval, and TechDetailsURL from metadata originally from the Data Package catalog["Contributor"] = tp.contributors[0]["title"] catalog["Vintage"] = tp.get_property("_vintage") catalog["UpdateInterval"] = tp.get_property("_updateInterval") + catalog["TechDetailsURL"] = tp.get_property("_techDetailsUrl") + + # Population MoreContextURL from metadata originally from the PresentationPackage + catalog["MoreContextURL"] = tp.get_property("_moreContextUrl") + + # Ensure the catalog has only the fields required by the schema and that they are in the correct order + catalog = catalog.filter(items=catalogSchema.field_names, axis="columns") + + # Ensure that the fields in the catalog are cast as the types specified in the schema + catalog = morpc.frictionless.cast_field_types(catalog, catalogSchema) + + # Sort the catalog according to the fields which comprise the primary key + catalog = catalog.sort_values(by=catalogSchema.primary_key) # Write the updated catalog to disk in the Tileset Package file structure logger.info("Writing updated catalog to disk") - sourcePath = os.path.normpath(os.path.join(tp.presentationPackageBasepath, catalogResource.path)) - destinationPath = os.path.normpath(os.path.join(tp.basepath, os.path.basename(catalogResource.path))) - if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): - logger.info(f"Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Existing catalog will be updated in-place.") - else: - logger.info(f"Destination path ({os.path.abspath(destinationPath)}) is different than source path ({os.path.abspath(sourcePath)}). Catalog provided with Presentation Package will be left unaltered.") + destinationPath = os.path.normpath(os.path.join(tp.basepath, "catalog.xlsx")) catalog.to_excel(destinationPath, index=False) + + # Compute the hash and filesize for the catalog on disk and update the resource in the + # Tileset Package + resource = tp.get_resource("catalog") + tp.remove_resource("catalog") + resource.hash = morpc.md5(destinationPath) + resource.bytes = os.path.getsize(destinationPath) + tp.add_resource(resource) # ### Validate tileset package logger.info("Checking completeness of Tileset Package object") @@ -528,43 +563,51 @@ def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", da # ## Create README.md file logger.info("Creating README.md file") - README_TEMPLATE = f''' - # {tp.title} + README_TEMPLATE = f'''# {tp.title} - ## Version +## Version - Current version: {tp.version} +Current version: {tp.version} - ## Contributors +## Contributors - {"\n".join([f"{x['title']}, {x['organization']}, {x['email']}" for x in tp.contributors])} +{"\n".join([f"{x['title']}, {x['organization']}, {x['email']}" for x in tp.contributors])} - ## Introduction +## Introduction - {tp.description} +{tp.description} - ## Data +## Data - Data file: {tp.get_resource("table").path} +Data file: {tp.get_resource("output").path} - Schema: {schemaPath} +Schema: {schemaPath} - ## Processes +## Processes - Process documentation: {tp.get_resource("process").path} - ''' +Process documentation: {tp.get_resource("process").path} - with open(os.path.join(tilesetBasepath, readmeResource.path), "w") as f: - f.write(README_TEMPLATE) +## Catalog + +The catalog spreadsheet includes links to visualizations of the data for each included geography as well as expert commentary for select geographies describing key insights suggested by the data. + +Catalog: {tp.get_resource("catalog").path} +''' + + if(produceReadme == True): + with open(os.path.join(tilesetBasepath, readmeResource.path), "w") as f: + f.write(README_TEMPLATE) # ## Write tileset package to disk and validate it logger.info(f"Writing tileset package descriptor to disk") destinationPath = os.path.normpath(os.path.join(tilesetBasepath, TILESET_PACKAGE_FILENAME)) tp.to_yaml(destinationPath); - results = morpc.frictionless.validate(destinationPath) + results = frictionless.validate(destinationPath) if(not results.valid): logger.error("Tileset Package is not a valid Frictionless Package. Details follow.") logger.error(results) + else: + logger.info("Tileset Package is a valid Frictionless Package.") # ## Clean up # Delete the misc directory if it is empty. Not all tilesets include misc content. Same with input data directory. From 8c9ea293980b1fce4d1c738714ba3d5c7dafc60e Mon Sep 17 00:00:00 2001 From: Adam Porr Date: Fri, 9 Oct 2026 11:11:03 -0400 Subject: [PATCH 5/5] Default to raise errors on all resource validation failures. Allow passing of keyword arguments to frictionless.Resource.validate. Previously morpc.frictionless.validate_resource only returned True or False and did not itself raise errors. Error determination was left to the code that called the function. Now validate_resource raises a RuntimeError by default, however this can be suppressed by setting new argument raiseErrors to False. validate_resource now accepts arbitrary keyword arguments, which are passed on to frictionless.Resource.validate. The primary reason for this is the need to override geometry checks when validating geopackages. morpc.frictionless.create_resource and morpc.frictionless.load_resource also accept validation keyword arguments. --- morpc/frictionless/frictionless.py | 36 +++++++++++++++++++++--------- morpc/frictionless/gpkg.py | 6 +++++ 2 files changed, 32 insertions(+), 10 deletions(-) diff --git a/morpc/frictionless/frictionless.py b/morpc/frictionless/frictionless.py index 7d3902a..9f518d1 100644 --- a/morpc/frictionless/frictionless.py +++ b/morpc/frictionless/frictionless.py @@ -562,7 +562,7 @@ def _verify_hash(path, expected): def create_resource(dataPath, title=None, name=None, description=None, sources=None, resourcePath=None, schemaPath=None, resFormat=None, resProfile=None, resMediaType=None, computeHash=True, computeBytes=True, ignoreSchema=False, writeResource=False, validate=False, control=None, lineEnds: Literal['dos', 'unix'] = 'dos', - cache=None, hashAlgorithm: Literal['md5', 'sha256'] = 'md5'): + cache=None, hashAlgorithm: Literal['md5', 'sha256'] = 'md5', **validateArgs): """Create a Frictionless resource object using sane default values for some attributes. Optionally, write the resource file to disk and validate the resource file, schema, and data. @@ -631,6 +631,10 @@ def create_resource(dataPath, title=None, name=None, description=None, sources=N Optional. The algorithm used to compute the hash attribute. Defaults to 'md5', which is emitted as a bare hex digest for backward compatibility (Data Package v1 style). 'sha256' is emitted in the self-describing Data Package v2 form "sha256:". + **validateArgs + This allows you to enter keyword arguments that will be passed through to frictionless.validate. For example, if + you include checkValidGeometry=False when calling this function, frictionless.validate will be called with + checkValidGeometry=False. Returns ------- @@ -834,16 +838,21 @@ def create_resource(dataPath, title=None, name=None, description=None, sources=N logger.info("Writing Frictionless Resource file to {}".format(resourceFilePath)) write_resource(resource, resourceFilePath) else: - logger.error("Unable to validate resource. No resource file path specified.") + logger.error("Unable to write resource. No resource file path specified.") raise RuntimeError if(validate == True): if(resourceFilePath != None): logger.info("Validating resource on disk.") - validate_resource(resourceFilePath) + resourceValid = validate_resource(resourceFilePath, **validateArgs) else: logger.error("Unable to validate resource. No resource file path specified.") - raise RuntimeError + raise RuntimeError + + if(not resourceValid): + logger.error("Validation failed. Errors should be described above.") + if(validateArgs.get("raiseErrors", None) != False): + raise RuntimeError return resource @@ -876,7 +885,7 @@ def write_resource(resource, resourcePath): os.chdir(cwd) -def validate_resource(resourcePath): +def validate_resource(resourcePath, raiseErrors=True, **kwargs): import os import frictionless @@ -890,13 +899,13 @@ def validate_resource(resourcePath): # different line endings than it was hashed with. For a local CSV, check those two ourselves. dataPath = resourceOnDisk.path if(isinstance(dataPath, str) and not _is_url(dataPath) and dataPath.lower().endswith(".csv") and os.path.exists(dataPath)): - results = resourceOnDisk.validate(checklist=frictionless.Checklist(skip_errors=["hash-count", "byte-count"])) + results = resourceOnDisk.validate(checklist=frictionless.Checklist(skip_errors=["hash-count", "byte-count"]), **kwargs) (algorithm, digest) = _parse_hash(resourceOnDisk.hash) if resourceOnDisk.hash else ('md5', None) integrityValid = any((digest == None or variantDigest == digest) and (resourceOnDisk.bytes == None or variantBytes == resourceOnDisk.bytes) for (variantDigest, variantBytes) in _line_end_variants(dataPath, algorithm)) else: - results = resourceOnDisk.validate() + results = resourceOnDisk.validate(**kwargs) integrityValid = True except Exception as e: @@ -912,6 +921,8 @@ def validate_resource(resourcePath): return True else: logger.error(f"Resource is NOT valid. Errors follow. {results}") + if(raiseErrors): + raise RuntimeError return False def _detect_sqlite_geometry_column(con, tableName): @@ -1021,7 +1032,7 @@ def resolve_data_path(resource, sourceDir, download=True): return os.path.join(sourceDir, resource.path) -def load_data(resourcePath, archiveDir=None, validate=False, forceInteger=False, forceInt64=False, useSchema="default", sheetName=None, layerName=None, tableName=None, driverName=None, targetCRS=None): +def load_data(resourcePath, archiveDir=None, validate=False, forceInteger=False, forceInt64=False, useSchema="default", sheetName=None, layerName=None, tableName=None, driverName=None, targetCRS=None, **validateArgs): """Often we want to make a copy of some input data and work with the copy, for example to protect the original data or to create an archival copy of it so that we can replicate the process later. The `load_data()` function simplifies the process of reading the data and @@ -1064,6 +1075,10 @@ def load_data(resourcePath, archiveDir=None, validate=False, forceInteger=False, Optional. The coordinate reference system to reproject the geometry to when loading a spatial SQLite database. Only used when a geometry column is detected in a SQLite file. SQLite WKB geometry carries no CRS information, so it is assumed to be "epsg:4326" on read. If None (the default), the data's native CRS is returned without reprojection. See morpc.load_spatial_data. + **validateArgs + This allows you to enter keyword arguments that will be passed through to frictionless.validate. For example, if + you include checkValidGeometry=False when calling this function, frictionless.validate will be called with + checkValidGeometry=False. Returns ------- @@ -1185,10 +1200,11 @@ def load_data(resourcePath, archiveDir=None, validate=False, forceInteger=False, if(validate): logger.info("Validating resource including data and schema (if applicable).") - resourceValid = validate_resource(targetResource) + resourceValid = validate_resource(targetResource, **validateArgs) if(not resourceValid): logger.error("Validation failed. Errors should be described above.") - raise RuntimeError + if(validateArgs.get("raiseErrors", None) != False): + raise RuntimeError logger.info("Loading data.") if(dataFileExtension == ".csv"): diff --git a/morpc/frictionless/gpkg.py b/morpc/frictionless/gpkg.py index dbfda99..927e460 100644 --- a/morpc/frictionless/gpkg.py +++ b/morpc/frictionless/gpkg.py @@ -199,6 +199,7 @@ def create_gpkgresource( validate=False, cache=None, hashAlgorithm="md5", + **validateArgs ): """Create one Frictionless Resource per layer of a GeoPackage file. @@ -228,6 +229,10 @@ def create_gpkgresource( resourceDir : str, optional Directory to write each layer's resource file to. Each file is named "{dataFileName}-{layerName}.resource.yaml". Required if writeResource is True. + **validateArgs + This allows you to enter keyword arguments that will be passed through to frictionless.validate. For example, if + you include checkValidGeometry=False when calling this function, frictionless.validate will be called with + checkValidGeometry=False. Returns ------- @@ -277,6 +282,7 @@ def create_gpkgresource( control=GpkgControl(layer=layerName), cache=cache, hashAlgorithm=hashAlgorithm, + **validateArgs ) resources.append(resource)