Skip to content

climate_ref.datasets.esmvaltool_reference #

Adapter for ESMValTool reference (observational/reanalysis) datasets.

ESMValTool reference data is not CMOR/obs4MIPs compliant, so metadata cannot be read from global attributes the way :mod:climate_ref.datasets.obs4mips does. Instead it is parsed from the ESMValCore DRS path and filename templates by :mod:climate_ref_core.esmvaltool_reference, which describes the layouts.

ESMValToolReferenceDatasetAdapter #

Bases: DatasetAdapter

Adapter for ESMValTool reference datasets.

See the module docstring for the layout conventions this adapter understands.

Source code in packages/climate-ref/src/climate_ref/datasets/esmvaltool_reference.py
class ESMValToolReferenceDatasetAdapter(DatasetAdapter):
    """
    Adapter for ESMValTool reference datasets.

    See the module docstring for the layout conventions this adapter understands.
    """

    dataset_cls: type[Dataset] = ESMValToolReferenceDataset
    slug_column = "instance_id"

    dataset_specific_metadata = (
        "project",
        "source_id",
        "variable_id",
        "frequency",
        "version",
        "data_type",
        "tier",
        "long_name",
        "units",
        "finalised",
        slug_column,
    )

    file_specific_metadata = ("start_time", "end_time", "path")
    version_metadata = "version"
    dataset_id_metadata = (
        "project",
        "source_id",
        "frequency",
        "variable_id",
    )

    def __init__(self, n_jobs: int = 1):
        self.n_jobs = n_jobs

    def find_local_datasets(self, file_or_directory: Path) -> pd.DataFrame:
        """
        Generate a data catalog from the specified file or directory.

        Each dataset may contain multiple files (rows). The unique dataset identifier is
        the ``instance_id`` slug in :attr:`slug_column`.
        """
        datasets = build_catalog(
            paths=[str(file_or_directory)],
            parsing_func=parse_esmvaltool_reference,
            include_patterns=["*.nc"],
            n_jobs=self.n_jobs,
        )
        if datasets.empty:
            logger.error("No datasets found")
            raise ValueError("No ESMValTool reference datasets found")

        datasets["start_time"] = parse_cftime_dates(datasets["start_time"])
        datasets["end_time"] = parse_cftime_dates(datasets["end_time"])
        datasets["finalised"] = True
        return build_instance_id(datasets, list(_INSTANCE_ID_FACETS), prefix=_SLUG_PREFIX)

find_local_datasets(file_or_directory) #

Generate a data catalog from the specified file or directory.

Each dataset may contain multiple files (rows). The unique dataset identifier is the instance_id slug in :attr:slug_column.

Source code in packages/climate-ref/src/climate_ref/datasets/esmvaltool_reference.py
def find_local_datasets(self, file_or_directory: Path) -> pd.DataFrame:
    """
    Generate a data catalog from the specified file or directory.

    Each dataset may contain multiple files (rows). The unique dataset identifier is
    the ``instance_id`` slug in :attr:`slug_column`.
    """
    datasets = build_catalog(
        paths=[str(file_or_directory)],
        parsing_func=parse_esmvaltool_reference,
        include_patterns=["*.nc"],
        n_jobs=self.n_jobs,
    )
    if datasets.empty:
        logger.error("No datasets found")
        raise ValueError("No ESMValTool reference datasets found")

    datasets["start_time"] = parse_cftime_dates(datasets["start_time"])
    datasets["end_time"] = parse_cftime_dates(datasets["end_time"])
    datasets["finalised"] = True
    return build_instance_id(datasets, list(_INSTANCE_ID_FACETS), prefix=_SLUG_PREFIX)

parse_esmvaltool_reference(file, **kwargs) #

Parse a single ESMValTool reference file into a metadata record.

Metadata comes from the path rather than the file contents, because the data is not CMOR compliant.

Source code in packages/climate-ref/src/climate_ref/datasets/esmvaltool_reference.py
def parse_esmvaltool_reference(file: str, **kwargs: Any) -> dict[str, Any]:
    """
    Parse a single ESMValTool reference file into a metadata record.

    Metadata comes from the path rather than the file contents,
    because the data is not CMOR compliant.
    """
    try:
        info = parse_reference_path(file)._asdict()
        timerange = info.pop("timerange")

        info["start_time"], info["end_time"] = parse_drs_daterange(timerange) if timerange else (None, None)
        info["path"] = str(file)
        info["long_name"] = None
        info["units"] = None
        return info
    except (ValueError, IndexError) as err:
        logger.warning(str(err))
        return {"INVALID_ASSET": file, "TRACEBACK": str(err)}
    except Exception:
        logger.warning(traceback.format_exc())
        return {"INVALID_ASSET": file, "TRACEBACK": traceback.format_exc()}