diff --git a/morpc/__init__.py b/morpc/__init__.py index c5566f9..9223f63 100644 --- a/morpc/__init__.py +++ b/morpc/__init__.py @@ -13,3 +13,4 @@ import morpc.rest_api import morpc.osm import morpc.utils +import morpc.insights diff --git a/morpc/insights/__init__.py b/morpc/insights/__init__.py new file mode 100644 index 0000000..e9113f7 --- /dev/null +++ b/morpc/insights/__init__.py @@ -0,0 +1 @@ +from .insights import * diff --git a/morpc/insights/insights.py b/morpc/insights/insights.py new file mode 100644 index 0000000..ea5e855 --- /dev/null +++ b/morpc/insights/insights.py @@ -0,0 +1,621 @@ +""" +Functions for assembling and validating tileset packages for the MORPC +Insights platform +Reference: https://github.com/morpc/morpc-insights +""" + +import logging +import frictionless + +logger = logging.getLogger(__name__) + +class DataPackage(frictionless.Package): + _requiredProps = ["resources","name","title","description","version","contributors","_vintage","_updateInterval","_techDetailsUrl"] + + def from_descriptor(self, descriptor): + import os + self.basepath = os.path.normpath(os.path.dirname(descriptor)) + package = super().from_descriptor(descriptor) + package.__class__ = DataPackage + return package + + def get_property(self, prop): + if(prop[0] == "_"): + return self.custom[prop] + else: + return getattr(self, prop) + + def set_property(self, prop, value): + if(prop[0] == "_"): + self.custom[prop] = value + else: + setattr(self, prop, value) + + def is_complete(self): + # Verify that required properties have been defined + definedProps = self.list_defined() + for key in self.custom.keys(): + definedProps.append(key) + missingProps = set(self._requiredProps).difference(set(definedProps)) + if(len(missingProps) > 0): + logger.error(f"The following required properties are not defined: {missingProps}") + return False + else: + return True + + complete = True + if(not "output" in self.resources): + logger.error(f"Resources must include one resource named 'output' which provides the metadata for the long-form data table") + complete = False + if(not "process" in self.resource_names): + logger.error(f"Resources must include one resource named 'process' which provides the metadata for a document describing the process by which the data was produced.") + complete = False + + return complete + + +class PresentationPackage(frictionless.Package): + _requiredProps = ["resources","name","title","description","version","contributors","_visualizationSpec","_visualizationSpecOverride","_moreContextUrl"] + + def from_descriptor(self, descriptor): + import os + self.basepath = os.path.normpath(os.path.dirname(descriptor)) + package = super().from_descriptor(descriptor) + package.__class__ = PresentationPackage + return package + + def get_property(self, prop): + if(prop[0] == "_"): + return self.custom[prop] + else: + return getattr(self, prop) + + def set_property(self, prop, value): + if(prop[0] == "_"): + self.custom[prop] = value + else: + setattr(self, prop, value) + + def is_complete(self): + # Verify that required properties have been defined + definedProps = self.list_defined() + for key in self.custom.keys(): + definedProps.append(key) + missingProps = set(self._requiredProps).difference(set(definedProps)) + if(len(missingProps) > 0): + logger.error(f"The following required properties are not defined: {missingProps}") + return False + else: + return True + + complete = True + if(not "catalog" in self.resource_names): + logger.error(f"Resources must include one resource named 'catalog' which provides a descriptor for the catalog table") + complete = False + + return complete + +class TilesetPackage(frictionless.Package): + _nativeProps = ["name","title","description","version"] + _requiredProps = ["resources","name","title","description","version","contributors","_vintage", + "_updateInterval","_techDetailsUrl","_visualizationSpec","_visualizationSpecOverride", + "_moreContextUrl"] + + def from_descriptor(self, descriptor): + import os + self.basepath = os.path.normpath(os.path.dirname(descriptor)) + package = super().from_descriptor(descriptor) + package.__class__ = PresentationPackage + return package + + def props_from_packages(self, dataPackage, presentationPackage): + self.resources = [] + self.dataPackageBasepath = dataPackage.basepath + self.presentationPackageBasepath = presentationPackage.basepath + for prop in dataPackage._requiredProps: + if((prop in self._nativeProps) or (prop == "resources")): + continue + else: + self.set_property(prop, dataPackage.get_property(prop)) + for prop in presentationPackage._requiredProps: + if((prop in self._nativeProps) or (prop == "resources")): + continue + else: + self.set_property(prop, presentationPackage.get_property(prop)) + + def get_property(self, prop): + if(prop[0] == "_"): + return self.custom[prop] + else: + return getattr(self, prop) + + def set_property(self, prop, value): + if(prop[0] == "_"): + self.custom[prop] = value + else: + setattr(self, prop, value) + + def is_complete(self): + # Verify that required properties have been defined + definedProps = self.list_defined() + for key in self.custom.keys(): + definedProps.append(key) + missingProps = set(self._requiredProps).difference(set(definedProps)) + if(len(missingProps) > 0): + logger.error(f"The following required properties are not defined: {missingProps}") + return False + + complete = True + if(not "catalog" in self.resource_names): + logger.error(f"Resources must include one resource named 'catalog' which provides the metadata for an Excel document which includes the details required for this tileset for inclusion in the Insights platform catalog") + complete = False + if(not "output" in self.resource_names): + logger.error(f"Resources must include one resource named 'table' which provides the metadata for the long-form data table") + complete = False + if(not "process" in self.resource_names): + logger.error(f"Resources must include one resource named 'process' which provides the metadata for a document describing the process by which the data was produced.") + complete = False + if(not "readme" in self.resource_names): + logger.error(f"Resources must include one resource named 'readme' which provides the metadata for a document (ideally named README.md and in Markdown format) which provides human-readable notes and metadata for this tileset.") + complete = False + if(not "maintainer" in self.resource_names): + logger.error(f"Resources must include one resource named 'maintainer' which provides the metadata for a document (ideally named MAINTAINER.md and in Markdown format) which describes the process to update the tileset.") + complete = False + + return complete + +def assemble_tileset_package(tilesetSlug, tilesetTitle, tilesetBasepath="./", dataPackagePath=None, presentationPackagePath=None, tilesetVersion=None, tilesetDescription=None, thumbnailUrlBase=None, produceReadme=False): + """ + Given an Insights Data Package and an Insights Presentation Package, create an Insights Tileset Package including the associated + Frictionless package file and all of metadata and artifacts provided in the two input packages. + + Parameters + ---------- + tilesetSlug : str + A short string that uniquely identifies the tileset. Typically the GitHub repository slug, e.g. "insights-pop". + tilesetTitle : str + The title for the tileset package. This will be combined with TITLE_PREFIX (see below) and the result will be + used for the Frictionless "title" property + tilesetBasepath : str + A string representing the path to the root directory for the tileset contents. This directory will be created if + it doesn't already exist. If no path is provided, the current working directory will be used. + dataPackagePath : str + A string representing the path to the Frictionless YAML file that defines an Insights Data Package. This should + be present in the root folder of the Data Package and should have the name "data.package.yaml". If no path is provided, + the script will look for "data.package.yaml" in in the directory specified for tilesetBasepath. + presentationPackagePath : str + A string representing the path to the Frictionless YAML file that defines an Insights Presentation Package. This should + be present in the root folder of the Presentation Package and should have the name "presentation.package.yaml". If no + path is provided, the script will look for "presentation.package.yaml" in in the directory specified for tilesetBasepath. + tilesetVersion : str + A string that uniquely identifies this version of the tileset. Typically using the format YYYY.mm.dd. If tilesetVersion + is not specified by the user, it will be constructed using the current date. + tilesetDescription : str + A brief description of the subject matter covered by the tileset. Will be used for the Frictionless "description" + property. If tilesetDescription is not specified by the user, the description from the Presentation Package will be + used. + thumbnailUrlBase : str + The base URL by which the thumbnail images will be accessed. Each thumbnail should be accessible by appending the thumbnail + filename to the base URL. Typically the URL would point to a GitHub repository. If thumbnailUrlBase is not specified, the script + will assume that GitHub is being used and will construct a URL using the tilesetSlug and the figures directory (see below) + produceReadme : bool + If set to True, the function will automatically produce a basic README.md file using the available metadata. By + default, no README.md file will be produced. + + Returns + ------- + tp : morpc.insights.TilesetPackage + An object representing an Insights Tileset Package. Includes metadata from the referenced Data Package and + Presentation Package + """ + # ## Setup + # ### Import required packages + import morpc + import frictionless + import datetime + import xlsxwriter + import pandas as pd + import os + import posixpath + import pathlib + import shutil + import logging + + # ### Static parameters + # Prefix that will be applied to the tileset title + TITLE_PREFIX = "MORPC Insights" + # Subdirectory of tileset directory where output data will be stored + OUTPUT_DATA_DIR = "output_data" + # Subdirectory of tileset directory where input data will be stored + INPUT_DATA_DIR = "input_data" + # Subdirectory of tileset directory where figures will be stored + FIGURES_DIR = "figures" + # Subdirectory of tileset directory where non-required will be stored + MISC_DIR = "misc" + # Default file name to use for data package if no path is provided + DEFAULT_DATAPACKAGE_FILENAME = "data.package.yaml" + # Default file name to use for presentation package if no path is provided + DEFAULT_PRESENTATIONPACKAGE_FILENAME = "presentation.package.yaml" + # Filename for the tileset package when written to disk + TILESET_PACKAGE_FILENAME = "tileset.package.yaml" + + # ### Start logging + logger = logging.getLogger(__name__) + + # ### Interpret arguments + if(dataPackagePath is None): + dataPackagePath = os.path.normpath(os.path.join("./", DEFAULT_DATAPACKAGE_FILENAME)) + logger.info(f"Data Package path was not specified. Using default path {dataPackagePath}") + if(presentationPackagePath is None): + presentationPackagePath = os.path.normpath(os.path.join("./", DEFAULT_PRESENTATIONPACKAGE_FILENAME)) + logger.info(f"Presentation Package path was not specified. Using default path {presentationPackagePath}") + if(tilesetVersion is None): + tilesetVersion = datetime.datetime.now().strftime("%Y.%m.%d") + logger.info(f"Tileset version was not specified. Using version derived from today's date {tilesetVersion}") + if(thumbnailUrlBase is None): + thumbnailUrlBase = f"https://raw.githubusercontent.com/morpc-insights/{tilesetSlug}/refs/heads/main/{FIGURES_DIR}/" + logger.info(f"Thumbnail base URL was not specified. Using URL constructed from the tileset slug and the figures directory name: {thumbnailUrlBase}") + + # ## Prepare inputs + # ### Read data package configuration + dp = morpc.insights.DataPackage().from_descriptor(dataPackagePath) + results = dp.validate() + if(not results.valid): + logger.error("Input Data Package is not a valid Frictionless Package. Details follow.") + logger.error(results) + raise RuntimeError + if(not dp.is_complete()): + logger.error("Input Data Package is missing one or more required elements.") + raise RuntimeError + else: + logger.info(f"Reading Data Package at {dataPackagePath}") + + # ### Read presentation package configuration + pp = morpc.insights.PresentationPackage().from_descriptor(presentationPackagePath) + results = pp.validate() + if(not results.valid): + logger.error("Input Presentation Package is not a valid Frictionless Package. Details follow.") + logger.error(results) + raise RuntimeError + if(not pp.is_complete()): + logger.error("Input Presentation Package is missing one or more required elements.") + raise RuntimeError + else: + logger.info(f"Reading Presentation Package at {presentationPackagePath}") + + if(tilesetDescription is None): + tilesetDescription = pp.description + logger.info(f"Tileset description was not specified. Using description provided in Presentation Package") + + # ## Create tileset package + # ### Create tileset directory structure + logger.info("Creating tileset directory structure (as needed)") + tilesetBasepath = os.path.normpath(tilesetBasepath) + tilesetOutputDataPath = os.path.join(tilesetBasepath, OUTPUT_DATA_DIR) + tilesetInputDataPath = os.path.join(tilesetBasepath, INPUT_DATA_DIR) + tilesetFiguresPath = os.path.join(tilesetBasepath, FIGURES_DIR) + tilesetMiscPath = os.path.join(tilesetBasepath, MISC_DIR) + if not os.path.exists(tilesetBasepath): + os.makedirs(tilesetBasepath) + if not os.path.exists(tilesetOutputDataPath): + os.makedirs(tilesetOutputDataPath) + if not os.path.exists(tilesetInputDataPath): + os.makedirs(tilesetInputDataPath) + if not os.path.exists(tilesetFiguresPath): + os.makedirs(tilesetFiguresPath) + if not os.path.exists(tilesetMiscPath): + os.makedirs(tilesetMiscPath) + + # ### Define tileset package properties + logger.info("Defining tileset package properties") + tp = morpc.insights.TilesetPackage() + tp.props_from_packages(dp, pp) + tp.basepath = tilesetBasepath + tp.set_property("name", tilesetSlug) + tp.set_property("title", tilesetTitle) + tp.set_property("description", tilesetDescription) + tp.set_property("version", tilesetVersion) + tp.dataPackageBasepath = posixpath.dirname(pathlib.PurePath(dataPackagePath).as_posix()) + tp.presentationPackageBasepath = posixpath.dirname(pathlib.PurePath(presentationPackagePath).as_posix()) + tp.resources = [] + + # ### Add resources to tileset package + # #### README + logger.info("Adding README.md resource to tileset package.") + readmeResource = frictionless.Resource() + readmeResource.name = "readme" + readmeResource.title= f"{TITLE_PREFIX} | {tilesetTitle} | README" + readmeResource.description = f"This document provides an overview of the {tilesetSlug} ({tilesetTitle}) tileset and its components." + readmeResource.path = "README.md" + tp.add_resource(readmeResource); + + # #### Maintainer guide + logger.info("Adding Maintainer Guide resource to tileset package.") + maintainerResource = frictionless.Resource() + maintainerResource.name = "maintainer" + maintainerResource.title= f"{TITLE_PREFIX} | {tilesetTitle} | Maintainer Guide" + maintainerResource.description = f"This document describes how to update the {tilesetSlug} ({tilesetTitle}) tileset." + maintainerResource.path = "MAINTAINER.md" + tp.add_resource(maintainerResource); + + # #### Resources from data package + # Add resources to tileset package descriptor. + logger.info("Adding resources defined in data package.") + for resource in dp.resources: + if(resource.name in tp.resource_names): + logger.error(f"Resource {resource.name} already defined in tileset package") + raise RuntimeError + else: + logger.info(f"--> Resource: {resource.name}") + if(resource.name.find("output") != -1): + # Put the primary output table (and schema and resource file if available) in the output data directory. If there are other outputs, do the same for those. + destinationDir = OUTPUT_DATA_DIR + elif(resource.name.find("process") != -1): + # Put any process definitions (i.e. any resources whose names include "process") in the root directory + destinationDir = "" + elif(resource.name.find("input") != -1): + # Put any input data (and schema and resource file if available) in the input data directory + destinationDir = INPUT_DATA_DIR + else: + # Put all other resources in the misc directory + destinationDir = MISC_DIR + newResource = resource.to_copy() + if("_cache" in resource.custom): + newResource.custom["_cache"] = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.custom["_cache"]).as_posix())) + else: + newResource.path = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.path).as_posix())) + if("schema" in resource.metadata_export()): + newResource.schema = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.metadata_export()["schema"]).as_posix())) + tp.add_resource(newResource) + + # Copy files specified as resources from the data package file structure to the tileset package file structure. For data files, attempt + # to copy the schema and resource files too, if available. + for resourceName in dp.resource_names: + logger.info(f"Copying files associated with Data Package resource '{resourceName}'") + resource = dp.get_resource(resourceName) + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, resource.path)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, resource.path)) + + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + + # For the long form table (required) the and any resources designated as inputs or outputs (optional), we also need to copy + # the schema and standalone resource file. The schema is required for the long form table, but optional for others. + if((resourceName == "table") or (resource.name.find("input") != -1) or (resource.name.find("output") != -1)): + # In the next line, I would think you could access the schema filename as resource.schema, + # however frictionless seems to dereference the schema automatically and it returns a + # schema object. It is possible to get the schema filename through metadata_export, but + # maybe there is a better way? + if("schema" in resource.metadata_export()): + schemaPath = resource.metadata_export()["schema"] + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, schemaPath)) + if(os.path.exists(sourcePath)): + logger.info(f"--> Detected schema associated with resource {resourceName}. Copying schema from data package to tileset package") + destinationPath = os.path.normpath(os.path.join(tp.basepath, schemaPath)) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + else: + logger.error(f"--> Schema defined for resource '{resource}' but specified file does not exist ({sourcePath})") + raise RuntimeError + else: + if(resourceName == "table"): + logger.error(f"--> Schema is required for resource 'table' but schema path was not specified.") + raise RuntimeError + else: + logger.warning(f"--> Schema path not specified for resource '{resourceName}'. Schema will not be included in Tileset Package.") + + # There may not be a standalone resource file because the resource information is required to be captured in the package file. + # Guess at the name of a standalone resource. If it exists, we'll grab that too. Our guess assumes that the resource file is located + # in the same directory as the data file and is named similarly but with extension ".resource.yaml" + extension = os.path.splitext(resource.path)[1] + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, resource.path.replace(extension,".resource.yaml"))) + if(os.path.exists(sourcePath)): + logger.info("--> Detected standalone resource file. Copying resource file from data package to tileset package") + destinationPath = os.path.normpath(os.path.join(tp.basepath, resource.path.replace(extension,".resource.yaml"))) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + + # #### Resources from presentation package + # Add resources to tileset package. + logger.info("Adding resources defined in presentation package.") + for resource in pp.resources: + if(resource.name in tp.resource_names): + logger.error(f"Resource {resource.name} already defined in tileset package") + raise RuntimeError + else: + logger.info(f"--> Resource: {resource.name}") + newResource = resource.to_copy() + if(resource.name == "commentary"): + # We'll use the commentary spreadsheet as the basis for the tileset catalog, which + # goes in the root directory + destinationDir = "" + newResource.path = posixpath.join(destinationDir, "catalog.xlsx") + newResource.schema = posixpath.join(destinationDir, "catalog.schema.yaml") + newResource.name = "catalog" + tp.add_resource(newResource) + else: + # Put all other resources in the misc direcotry + destinationDir = posixpath.join(tp.basepath, MISC_DIR) + newResource.path = posixpath.join(destinationDir, posixpath.basename(pathlib.PurePath(resource.path).as_posix())) + tp.add_resource(newResource) + + # Copy files specified as resources from the presentation package file structure to the tileset package file structure. + resource_names = pp.resource_names + # We will not copy the commentary. Rather, we'll read it, make some adjustments, and write the adjusted version + # as the tileset catalog + resource_names.remove("commentary") + for resourceName in pp.resource_names: + logger.info(f"Copying files associated with Presentation Package resource '{resourceName}'") + resource = pp.get_resource(resourceName) + sourcePath = os.path.normpath(os.path.join(tp.dataPackageBasepath, resource.path)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, resource.path)) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + logger.info(f"--> Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file.") + else: + logger.info(f"--> Copying file from {sourcePath} to {destinationPath}") + shutil.copyfile(sourcePath, destinationPath) + + # ## Create catalog + logger.info("Loading commentary provided with presentation package") + commentaryResource = pp.get_resource("commentary") + commentary = pd.read_excel(commentaryResource.path) + + # Ensure that all values expected in the Presentation Package have been populated. + missingFlag = False + for column in ["GeoType","GeoName","Headline","Commentary","ThumbnailURL","DataProductURL"]: + if(not commentary.loc[commentary[column].isna()].empty): + missingFlag = True + if(column == "Headline"): + additionalInstructions = "The headlines for community-level geographies may be identical." + elif(column == "Commentary"): + additionalInstructions = "The commentary for community-level geographies may be identical." + elif(column == "ThumbnailURL"): + additionalInstructions = "Please enter URLs or paths for all geographies. If a local path is provided it will be converted to a URL using the thumbnailUrlBase value." + elif(column == "DataProductURL"): + additionalInstructions = "Please enter a data product URLs for all geographies. Each geography may have a unique URL, or any set of geographies may share a common URL." + logger.error(f"Blank values detected in column {column}. Please enter a value for for all geographies. {additionalInstructions}") + if(missingFlag == True): + raise RuntimeError + else: + logger.info("No missing values detected.") + + # Update thumbnail URLs and copy provided images from the presentation package to the tileset package if necessary + visualizationSpec = pp.get_property("_visualizationSpec") + if(visualizationSpec == "useProvided"): + logger.info("Creating copies of the thumbnails in the tileset package.") + filenames = commentary["ThumbnailURL"].apply(lambda x:os.path.basename(x)) + commentary["ThumbnailURL"] = thumbnailUrlBase + filenames + showNotification = True + for file in filenames: + sourcePath = os.path.normpath(os.path.join(tp.presentationPackageBasepath, FIGURES_DIR, file)) + destinationPath = os.path.normpath(os.path.join(tp.basepath, FIGURES_DIR, file)) + if(os.path.abspath(sourcePath) == os.path.abspath(destinationPath)): + if(showNotification == True): + logger.info(f"Destination path and source path resolve to the same absolute path ({os.path.abspath(sourcePath)}). Will not copy file. Notifications will be suppressed for remaining files.") + showNotification = False + else: + shutil.copyfile(sourcePath, destinationPath) + else: + # Eventually we will support construction of figures from a visualization specification. As of September 2026, + # this is not implemented. Return an error. + logger.error("Only visualizationSpec values useProvidedCopy and useProvidedInPlace are currently supported.") + raise RuntimeError + + # Create the tileset catalog using the commentary spreadsheet as the base + catalog = commentary.copy() + catalogSchema = tp.get_resource("catalog").schema + + # Create new fields TileID, TilesetID, Category, and ShareURL and leave them blank. These are managed + # by a downstream workflow. + catalog["TileID"] = None + catalog["TilesetID"] = None + catalog["Category"] = None + catalog["Priority"] = None + + # Populate Contributor, Vintage, UpdateInterval, and TechDetailsURL from metadata originally from the Data Package + catalog["Contributor"] = tp.contributors[0]["title"] + catalog["Vintage"] = tp.get_property("_vintage") + catalog["UpdateInterval"] = tp.get_property("_updateInterval") + catalog["TechDetailsURL"] = tp.get_property("_techDetailsUrl") + + # Population MoreContextURL from metadata originally from the PresentationPackage + catalog["MoreContextURL"] = tp.get_property("_moreContextUrl") + + # Ensure the catalog has only the fields required by the schema and that they are in the correct order + catalog = catalog.filter(items=catalogSchema.field_names, axis="columns") + + # Ensure that the fields in the catalog are cast as the types specified in the schema + catalog = morpc.frictionless.cast_field_types(catalog, catalogSchema) + + # Sort the catalog according to the fields which comprise the primary key + catalog = catalog.sort_values(by=catalogSchema.primary_key) + + # Write the updated catalog to disk in the Tileset Package file structure + logger.info("Writing updated catalog to disk") + destinationPath = os.path.normpath(os.path.join(tp.basepath, "catalog.xlsx")) + catalog.to_excel(destinationPath, index=False) + + # Compute the hash and filesize for the catalog on disk and update the resource in the + # Tileset Package + resource = tp.get_resource("catalog") + tp.remove_resource("catalog") + resource.hash = morpc.md5(destinationPath) + resource.bytes = os.path.getsize(destinationPath) + tp.add_resource(resource) + + # ### Validate tileset package + logger.info("Checking completeness of Tileset Package object") + if(not tp.is_complete()): + logger.error("--> Tileset package object is missing one or more required elements") + raise RuntimeError + else: + logger.info("--> Tileset package object is complete") + + # ## Create README.md file + logger.info("Creating README.md file") + + README_TEMPLATE = f'''# {tp.title} + +## Version + +Current version: {tp.version} + +## Contributors + +{"\n".join([f"{x['title']}, {x['organization']}, {x['email']}" for x in tp.contributors])} + +## Introduction + +{tp.description} + +## Data + +Data file: {tp.get_resource("output").path} + +Schema: {schemaPath} + +## Processes + +Process documentation: {tp.get_resource("process").path} + +## Catalog + +The catalog spreadsheet includes links to visualizations of the data for each included geography as well as expert commentary for select geographies describing key insights suggested by the data. + +Catalog: {tp.get_resource("catalog").path} +''' + + if(produceReadme == True): + with open(os.path.join(tilesetBasepath, readmeResource.path), "w") as f: + f.write(README_TEMPLATE) + + # ## Write tileset package to disk and validate it + logger.info(f"Writing tileset package descriptor to disk") + destinationPath = os.path.normpath(os.path.join(tilesetBasepath, TILESET_PACKAGE_FILENAME)) + tp.to_yaml(destinationPath); + results = frictionless.validate(destinationPath) + if(not results.valid): + logger.error("Tileset Package is not a valid Frictionless Package. Details follow.") + logger.error(results) + else: + logger.info("Tileset Package is a valid Frictionless Package.") + + # ## Clean up + # Delete the misc directory if it is empty. Not all tilesets include misc content. Same with input data directory. + if(len(os.listdir(tilesetMiscPath)) == 0): + os.rmdir(tilesetMiscPath) + if(len(os.listdir(tilesetInputDataPath)) == 0): + os.rmdir(tilesetInputDataPath) + + return tp + + \ No newline at end of file diff --git a/morpc/insights/insights_assemble_tileset_package.py b/morpc/insights/insights_assemble_tileset_package.py new file mode 100644 index 0000000..92c9614 --- /dev/null +++ b/morpc/insights/insights_assemble_tileset_package.py @@ -0,0 +1,89 @@ +# insights_assemble_tileset_package.py +# +# This script is a wrapper for morpc.insights.assemble_tileset_package(). Given an Insights Data Package +# and an Insights Presentation Package, create an Insights Tileset Package including the associated +# Frictionless package file and all of metadata and artifacts provided in the two input packages. + +import morpc +import logging +import argparse +import sys +import os + +LOGLEVEL = "info" + +LOGFORMAT = '%(asctime)s | %(levelname)s | %(name)s.%(funcName)s: %(message)s' + +LEVEL_MAP = { + "debug": 10, + "info": 20, + "warning": 30, + "error": 40, + "critical": 50 +} + +logging.basicConfig( + level=LEVEL_MAP[LOGLEVEL], + force=True, + format=LOGFORMAT, + handlers=[ + logging.StreamHandler(sys.stdout) + ] + ) +logging.getLogger(__name__).setLevel(LEVEL_MAP[LOGLEVEL]) + +epilogString = ''' +# Examples + +## Create tileset package in-situ (most common, least complex) + +Typically, you\'ll already have a GitHub repository which contains the Data Package elements and Presentation Package +elements and you just want to create the tileset Frictionless metadata (tileset.package.yaml) and update metadata in +the catalog. In that case, you can rely on the defaults for most arguments. + +NOTE: Your catalog.xlsx will be overwritten in this case and some values will be altered. + +python insights_assemble_tileset_package.py "insights-pop" "Historic and Forecasted Population by Year" "C:\\Users\\yourname\\github\\insights-pop" + +## Create separate tileset package in new folder using all optional arguments (for reference only, real use cases are uncommon) + +python insights_assemble_tileset_package.py "insights-pop" "Historic and Forecasted Population by Year" "C:\\Users\\yourname\\github\\insights-pop" --dataPackagePath "C:\\some_path_to\\data.package.yaml" --presentationPackagePath "C:\\another_path_to\\presentation.package.yaml" --tilesetVersion "2026.09.17" --tilesetDescription "I didn\'t like the description in the Presentation Package so I wrote this one instead" --thumbnailUrlBase "https://www.myserver.com/somedir/" +''' + +parser = argparse.ArgumentParser( + description = "This script is a wrapper for morpc.insights.assemble_tileset_package(). Given an Insights Data Package and an Insights Presentation Package, create an Insights Tileset Package including the associated Frictionless package file and all of metadata and artifacts provided in the two input packages.", + epilog = epilogString.replace("\\\\","\\"), + formatter_class=argparse.RawDescriptionHelpFormatter +) +parser.add_argument("tilesetSlug", + help = "A short string that uniquely identifies the tileset. Typically the GitHub repository slug, e.g. 'insights-pop'." +) +parser.add_argument("tilesetTitle", + help = "The title for the tileset package. This will be combined with TITLE_PREFIX (see morpc.insights code) and the result will be used for the Frictionless 'title' property") +parser.add_argument("tilesetBasepath", + help = "A string representing the path to the root directory for the tileset contents. This directory will be created if it doesn't already exist. If no path is provided, the current working directory will be used." +) +parser.add_argument("--dataPackagePath", + help = "A string representing the path to the Frictionless YAML file that defines an Insights Data Package. This should be present in the root folder of the Data Package and should have the name 'data.package.yaml'. If no path is provided, the script will look for 'data.package.yaml' in in the directory specified for tilesetBasepath." +) +parser.add_argument("--presentationPackagePath", + help = "A string representing the path to the Frictionless YAML file that defines an Insights Presentation Package. This should be present in the root folder of the Presentation Package and should have the name 'presentation.package.yaml'. If no path is provided, the script will look for 'presentation.package.yaml' in in the directory specified for tilesetBasepath." +) +parser.add_argument("--tilesetVersion", + help = "A string that uniquely identifies this version of the tileset. Typically using the format YYYY.mm.dd. If tilesetVersion is not specified by the user, it will be constructed using the current date." +) +parser.add_argument("--tilesetDescription", + help = "A brief description of the subject matter covered by the tileset. Will be used for the Frictionless 'description' property. If tilesetDescription is not specified by the user, the description from the Presentation Package will be used." +) +parser.add_argument("--thumbnailUrlBase", + help = "The base URL by which the thumbnail images will be accessed. Each thumbnail should be accessible by appending the thumbnail filename to the base URL. Typically the URL would point to a GitHub repository. If thumbnailUrlBase is not specified, the script will assume that GitHub is being used and will construct a URL using the tilesetSlug and the figures directory (see morpc.insights code)" +) +args = parser.parse_args() + +morpc.insights.assemble_tileset_package(args.tilesetSlug, args.tilesetTitle, os.path.normpath(args.tilesetBasepath), + dataPackagePath = args.dataPackagePath, + presentationPackagePath = args.presentationPackagePath, + tilesetVersion = args.tilesetVersion, + tilesetDescription = args.tilesetDescription, + thumbnailUrlBase = args.thumbnailUrlBase +) diff --git a/morpc/morpc.py b/morpc/morpc.py index c244006..b8b0608 100644 --- a/morpc/morpc.py +++ b/morpc/morpc.py @@ -871,6 +871,15 @@ "idField":"REGIONSWACOID", "nameField":"REGIONSWACO", "censusQueryName": None + }, + 'M31': { + "singular":"library district", + "plural":"library districts", + "hierarchy_string":"LIBRARYD", + "authority":"morpc", + "idField":"LIBRARYDID", + "nameField":"LIBRARYD", + "censusQueryName": None }, } @@ -2269,7 +2278,7 @@ def recursiveUpdate(original, updates): original[key] = value return original -def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dataOptions=None, chartOptions=None): +def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dataOptions=None, chartOptions=None, na_rep=None): # TODO: simplify docstring """ Create an Excel worksheet consisting of the contents of a pandas dataframe (as a formatted table) @@ -2414,6 +2423,9 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat "y2AxisOptions": dict Options to control the appearance of the secondary y axis. Will be used directly by chart.set_y2_axis(). No defaults are applied. All required values must by specified. See https://xlsxwriter.readthedocs.io/chart.html#chart-set-y2-axis + na_rep: str + Value to use represent null values in the data frame. Default is ''. Use '=NA()' to use Excel-compatible + null values. This is necessary when including null values in ranges to be shown in a line chart. Returns ------- @@ -2516,11 +2528,13 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat }, "smooth": False }) + seriesOptionsDefault["scatter"] = json.loads(json.dumps(seriesOptionsDefault["line"])) subtypesDefaults = { "bar": None, "column": None, - "line": None + "line": None, + "scatter": None } myDataOptions = { @@ -2574,7 +2588,7 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat workbook = writer.book - df.to_excel(writer, sheet_name=sheet_name, index=myDataOptions["index"]) + df.to_excel(writer, sheet_name=sheet_name, index=myDataOptions["index"], na_rep=na_rep) worksheet = writer.sheets[sheet_name] @@ -2668,7 +2682,6 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat color = myChartOptions["colors"][(i-1) % len(myChartOptions["colors"])] elif(type(myChartOptions["colors"]) == dict): color = myChartOptions["colors"].get(colname, styleDefaults["seriesColor"]) # Revert to default if color is not specified for column - json.dumps(mySeriesOptions, indent=4) # Else if we have more than one series, cycle through the default set of colors elif(nColumns > 1): color = colorsDefault[(i-1) % len(colorsDefault)] @@ -2732,7 +2745,7 @@ def data_chart_to_excel(df, writer, sheet_name="Sheet1", chartType="column", dat # the y-axis title was not provided in the dict, revert to the default. if(type(myChartOptions["titles"]) == dict): myYAxisOptions["name"] = myChartOptions["titles"].get("yTitle", yAxisOptionsDefaults["name"]) - + if(colname in y2AxisColumns): mySeriesOptions["y2_axis"] = True chart.add_series(mySeriesOptions)