Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions ci/isschecker_env_floor.yml
Original file line number Diff line number Diff line change
Expand Up @@ -24,5 +24,6 @@ dependencies:
- xarray =2025.1.2
- cftime =1.6.4
- netcdf4 =1.7
- cf-units =3.3

- pytest =8
1 change: 1 addition & 0 deletions docs/dev/source-install.md
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,7 @@ results should agree across machines and operating systems within these bounds.
| `xarray` | `>=2025.1.2,<2027` | `xarray.coders.CFDatetimeCoder` (public API in 2025.1.1) and non-nanosecond datetime decoding, both used by the time checks |
| `cftime` | `>=1.6.4,<2` | date arithmetic in the start/end/duration checks |
| `netCDF4` | `>=1.7,<2` | `_FillValue` checks compare against `netCDF4.default_fillvals` |
| `cf-units` | `>=3.3,<4` | UDUNITS-2, which decides whether a units attribute means what the data request asks for; 3.3 is the first release built against numpy 2 |
| `tqdm` | `>=4.66` | progress bar only; never affects the log |

If you report a problem with the checker, please include the output of
Expand Down
4 changes: 3 additions & 1 deletion docs/user/checks.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,9 @@ ancillary_variables). Anything further is a warning.
## 2. Numerical

Units match the data request in any UDUNITS spelling: m2, m^2 and m**2 are
all accepted, as are kg m-2 s-1, kg.m-2.s-1 and kg/m2/s. Every value is
all accepted, as are kg m-2 s-1, kg.m-2.s-1 and kg/m2/s, and K and kelvin.
The comparison is UDUNITS's own, so a string it cannot parse is an error, and
so is a unit at another scale, such as m yr-1 for m s-1. Every value is
either a finite number or the declared _FillValue, so a bare NaN is never how
a file says "missing". Values lie within the range allowed for the region: a
few outside it are a warning, a large share of the field is an error. The
Expand Down
112 changes: 28 additions & 84 deletions isschecker/checker.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,8 @@
#
# 2. Numerical (_check_numerical)
# - Variable units match the data request (any UDUNITS spelling of the
# requested unit is accepted: 'm2', 'm^2' and 'm**2' are all the same).
# requested unit is accepted: 'm2', 'm^2' and 'm**2' are all the same,
# as are 'K' and 'kelvin'). A string UDUNITS cannot parse is an error.
# - Every value is either a finite number or exactly the variable's declared
# _FillValue. A bare NaN or an infinity is a private spelling of
# "missing" that a reader filtering on _FillValue takes for data, so it is
Expand Down Expand Up @@ -140,6 +141,7 @@
from importlib import metadata, resources
from typing import NamedTuple

import cf_units
import numpy as np
import pandas as pd
import xarray as xr
Expand Down Expand Up @@ -1693,95 +1695,32 @@ def _check_file_variables(
return True


# CF requires only that the units attribute be "a string that can be recognized
# by the UDUNITS package", and UDUNITS recognises several spellings of the same
# unit: the exponent in 'm2' may equally be written 'm^2' or 'm**2', the factors
# of a product may be separated by a space, a '.', a '*' or a middle dot, and a
# '/' introduces factors with negated exponents. The data request writes one
# spelling per variable, but a model that writes another is just as compliant
# (PISM writes 'm^2' because pint does not recognise 'm2'), so the checker
# compares what units mean rather than how they are spelled.

# One factor of a product: a base name, then an exponent written with a caret
# ('m^2'), a double star ('m**2', normalised to a caret first) or bare ('m2').
# Digits and signs are kept out of the base so that the exponent is what is
# left over.
_UNIT_FACTOR_RE = re.compile(r"^(?P<base>[^\d\s^/*.()+-]+)\^?(?P<exponent>[+-]?\d+)?$")

# Whitespace, '.', '*' and the middle dot all multiply in UDUNITS.
_UNIT_PRODUCT_SEPARATORS = r"[\s.*·×]+"
_UNIT_PRODUCT_SEPARATOR_RE = re.compile(_UNIT_PRODUCT_SEPARATORS)

# Splits a units string into factors and the operators between them, keeping
# the operators so that '/' can be told from the multiplications.
_UNIT_TOKEN_RE = re.compile(rf"(/|{_UNIT_PRODUCT_SEPARATORS})")


def _parse_units(units: str):
"""Return a canonical (scale, factors) form of a UDUNITS string.

`factors` is the sorted tuple of (base unit, exponent) pairs, so that
equivalent spellings of the same unit give equal results. Returns None for
the strings this deliberately small parser does not claim to understand:
empty units, parentheses, timestamp offsets such as 'days since
1850-01-01', and decimal scale factors (where '.' is a decimal point rather
than a multiplication). Callers fall back to comparing such strings
literally.
"""
text = units.strip()
if (
not text
or "(" in text
or ")" in text
or re.search(r"\d\.\d", text)
or re.search(r"\b(since|after|from|ref)\b", text)
):
return None

text = text.replace("**", "^")
scale = 1.0
exponents: dict[str, int] = {}
# UDUNITS multiplies and divides left to right, so a '/' inverts only the
# factor that follows it: 'kg/m2 s' is kg m-2 s, not kg m-2 s-1.
sign = 1
factor_count = 0
for token in _UNIT_TOKEN_RE.split(text):
if not token:
continue
if token == "/":
sign = -1
continue
if _UNIT_PRODUCT_SEPARATOR_RE.fullmatch(token):
sign = 1
continue
match = _UNIT_FACTOR_RE.match(token)
if match is None:
# Not a named unit: the only other thing it can be is a numerical
# factor, such as the '1' of a dimensionless unit.
try:
scale *= float(token) ** sign
except ValueError:
return None
else:
base = match.group("base")
exponent = int(match.group("exponent") or 1) * sign
exponents[base] = exponents.get(base, 0) + exponent
sign = 1
factor_count += 1

if factor_count == 0:
def _unit(text: str):
"""The UDUNITS unit a string names, or None if UDUNITS cannot parse it."""
try:
return cf_units.Unit(text)
except ValueError:
return None

factors = tuple(sorted((b, e) for b, e in exponents.items() if e != 0))
return scale, factors


def _units_match(actual: str, expected: str) -> bool:
"""Whether two units strings denote the same unit, however each is spelled."""
"""Whether two units strings denote the same unit, however each is spelled.

CF requires only that the units attribute be "a string that can be
recognized by the UDUNITS package", and UDUNITS recognizes many spellings
of one unit: 'm2', 'm^2' and 'm**2'; 'kg m-2 s-1', 'kg/m2/s' and
'kg.m-2.s-1'; 'K' and 'kelvin'; 's' and 'second'. The data request writes
one spelling per variable, but a model that writes another is just as
compliant -- PISM writes the names its own UDUNITS gives it -- so the
comparison is UDUNITS's own, through cf-units. Equal means the same
dimension at the same scale with the same offset: 'm yr-1' is not
'm s-1', 'kPa' is not 'Pa' and 'degC' is not 'K'. A string UDUNITS cannot
parse matches nothing.
"""
if actual == expected:
return True
parsed_actual = _parse_units(actual)
return parsed_actual is not None and parsed_actual == _parse_units(expected)
actual_unit = _unit(actual)
return actual_unit is not None and actual_unit == _unit(expected)


def _fill_value(ds, ivar):
Expand Down Expand Up @@ -1914,6 +1853,11 @@ def _check_numerical(
+ expected_units
+ ")"
)
elif _unit(var_units) is None:
reporter.error(
f"The unit of the variable, '{var_units}', is not one UDUNITS"
f" recognizes, and should be {expected_units}"
)
else:
reporter.error(
"The unit of the variable is "
Expand Down
3 changes: 3 additions & 0 deletions isschecker_env.yml
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,9 @@ dependencies:
- xarray >=2025.1.2,<2027
- cftime >=1.6.4,<2
- netcdf4 >=1.7,<2
# cf-units is UDUNITS-2, which is what CF says a units attribute must be
# recognized by; >=3.3 is the first release built against numpy 2
- cf-units >=3.3,<4

# Testing
- pytest >=8
Expand Down
3 changes: 2 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"

[project]
name = "isschecker"
version = "0.4.0"
version = "0.5.0"
description = "ISMIP7 compliance checker for ice sheet model simulation NetCDF datasets"
readme = "README.md"
license = "MIT"
Expand All @@ -26,6 +26,7 @@ dependencies = [
"xarray>=2025.1.2,<2027",
"cftime>=1.6.4,<2",
"netCDF4>=1.7,<2",
"cf-units>=3.3,<4",
"tqdm>=4.66",
]

Expand Down
55 changes: 52 additions & 3 deletions tests/test_compliance_checker.py
Original file line number Diff line number Diff line change
Expand Up @@ -788,6 +788,10 @@ def test_checker_reports_flux_variable_with_state_timestamps(case_dir):
("tendacabf", "kg s^-1"),
("tendacabf", "kg/s"),
("tendacabf", "kg.s-1"),
# UDUNITS names as well as symbols: what PISM writes (discussion
# ismip#46).
("tendacabf", "kilogram second^-1"),
("tendacabf", "kg/second"),
],
)
def test_checker_accepts_equivalent_units_spellings(case_dir, variable_name, units):
Expand All @@ -812,6 +816,23 @@ def test_checker_reports_wrong_units(case_dir):
assert "The unit of the variable is m^3 and should be m^2" in summary["log_text"]


def test_checker_reports_a_units_string_udunits_cannot_parse(case_dir):
"""A string that is not a unit at all is its own finding.

'M' is not a spelling of metres, and saying so is more use than saying
it is the wrong unit, which would send the modeler looking for a factor.
"""
set_variable_units(dataset_for_variable(case_dir, "lim"), "M")

summary = run_checker(case_dir)

assert summary["total_num_errors"] == 1
assert (
"The unit of the variable, 'M', is not one UDUNITS recognizes, and"
" should be kg" in summary["log_text"]
)


@pytest.mark.parametrize(
"actual, expected, matches",
[
Expand All @@ -822,17 +843,31 @@ def test_checker_reports_wrong_units(case_dir):
("kg/m2/s", "kg m-2 s-1", True),
("s-1 kg m-2", "kg m-2 s-1", True),
("kg m-2 s-1", "kg m-2 s-1", True),
("kg/(m2 s)", "kg m-2 s-1", True),
("1", "1", True),
# Names as well as symbols, which is what PISM writes (discussion
# ismip#46).
("kelvin", "K", True),
("Kelvin", "K", True),
("degK", "K", True),
("kg m^-2 second^-1", "kg m-2 s-1", True),
("meters/second", "m s-1", True),
("N m-2", "Pa", True),
("m3", "m2", False),
("kg m-2", "kg m-2 s-1", False),
# Same dimension is not the same unit: the numbers would be off by a
# factor, or an offset, that the units attribute hides.
("km", "m", False),
("M", "m", False),
("kPa", "Pa", False),
("m yr-1", "m s-1", False),
("degC", "K", False),
("percent", "1", False),
# UDUNITS divides left to right, so the '/' inverts only 'm2' here.
("kg/m2*s", "kg m-2 s-1", False),
("kg/m2 s", "kg m-2 s-1", False),
# Not understood, so compared as strings rather than guessed at.
# Not a unit UDUNITS recognizes, so it matches nothing.
("M", "m", False),
("", "1", False),
("kg/(m2 s)", "kg m-2 s-1", False),
("days since 1850-01-01", "days since 1850-01-01", True),
("days since 1850-01-01", "days since 2000-01-01", False),
],
Expand All @@ -841,6 +876,20 @@ def test_units_match(actual, expected, matches):
assert checker._units_match(actual, expected) is matches


def test_every_requested_unit_is_one_udunits_recognizes():
"""A typo in the data request would otherwise fail every file of a variable.

The comparison is UDUNITS's own, so the request has to speak UDUNITS too.
"""
ismip_meta, _, _, _, _ = checker._load_criteria("ismip7")

assert {
entry["variable"]: entry["units"]
for entry in ismip_meta
if checker._unit(entry["units"]) is None
} == {}


def test_checker_reports_variable_missing_from_file(case_dir):
rename_variable_in_file(dataset_for_variable(case_dir, "lim"), "lim", "limm")

Expand Down
Loading