diff --git a/.hooks/check_netcdf4_imports.py b/.hooks/check_netcdf4_imports.py index 00cb1ef7dd..7fe43b30c0 100644 --- a/.hooks/check_netcdf4_imports.py +++ b/.hooks/check_netcdf4_imports.py @@ -30,6 +30,13 @@ "iris/tests/unit/fileformats/netcdf/_thread_safe_nc/test_NetCDFWriteProxy.py", # The tests for the bytecoding dataset wrapper. "iris/tests/unit/fileformats/netcdf/test_bytecoding_datasets.py", + # The test of NetCDFDataset.from_existing() wrapping a bare netCDF4.Dataset, + # as the Xarray bridge hands to iris.save / CFReader. + "iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py", + # The test that CFReader wraps a bare netCDF4.Dataset borrowed from a + # caller, decoding its character data - the same wrapping as above, but + # exercised through CFReader rather than NetCDFDataset directly. + "iris/tests/unit/fileformats/cf/test_CFReader__dataset.py", # The system test that checks netCDF4 is importable. "iris/tests/system_test.py", ) diff --git a/docs/superpowers/plans/2026-09-24-zarr-io-pr2.md b/docs/superpowers/plans/2026-09-24-zarr-io-pr2.md new file mode 100644 index 0000000000..4c469ed083 --- /dev/null +++ b/docs/superpowers/plans/2026-09-24-zarr-io-pr2.md @@ -0,0 +1,8274 @@ +# PR 2 — `CFDataset`, and the CF variable classes rewritten against it + +> **For agentic workers:** REQUIRED SUB-SKILL: Use +> `superpowers:subagent-driven-development` (recommended) or +> `superpowers:executing-plans` to implement this plan task-by-task. Steps use +> checkbox (`- [ ]`) syntax for tracking. + +> **This plan ships with its implementation.** Plan and code are delivered on +> the same branch and in the same pull request. If implementation proves a step +> wrong, correct the step here on the branch — the document is frozen when the +> work merges, not while it is under way. Programme status lives in §12 of the +> [design spec](../specs/2026-09-21-zarr-io-design.md); nothing here duplicates +> it. + +| | | +|---|---| +| **Spec** | [`../specs/2026-09-21-zarr-io-design.md`](../specs/2026-09-21-zarr-io-design.md) §4.2, §4.3, §4.4, §4.5, §5 (PR 2), §6 | +| **Implements** | The second of the seven pull requests in spec §5 | +| **Branch** | `zarr-io-pr-2`, targeting `SciTools/iris:brownfield` | +| **Baseline** | `brownfield` at `2cc0b01d4`, Iris `3.17.0.dev16`, `iris-test-data` at `f4a6a05`. Full suite green: `11724 passed, 66 skipped, 0 failed, 0 errors` (§6.2) | +| **Delivers** | `cf/dataset.py`, `netcdf/_dataset.py`, and every CF attribute and netCDF-API access in `cf/`, `_nc_load_rules/` and `netcdf/` routed through them | +| **Behaviour change** | Two, both about names that collide with a Python member: such an attribute now reaches the cube on load (F1, F13) and reaches the file on save (F13). **Not** the `spans` fix spec §5 anticipated — that gap turned out to be unreachable, so Task 7 characterises it instead (F9). Plus one new deprecation warning, on a path library code no longer takes | +| **Public API change** | Additive only: the new public module `iris.fileformats.cf.dataset` (`CFDataset`, `CFDatasetVariable`, `TrackedAttributes`) and `CFVariable.attributes` | + +**Created:** 2026-09-24 + +Line references are accurate at the baseline commit. Claims marked +**[verified]** were checked by running code or `git` against this working tree; +everything else is read from source. + +**Goal:** Give Iris a narrow, format-agnostic description of a CF-conforming +array store — `CFDataset` / `CFDatasetVariable` — implement it once for netCDF, +and rewrite every CF attribute read and every netCDF-only API call in the `cf` +package, `_nc_load_rules/` and `netcdf/` to go through it, so that PR 4 can add +a Zarr implementation and reuse the whole machine. + +**Architecture:** Two abstract classes in `cf/dataset.py` declare what the CF +layer needs from a store: typed properties (`dimensions`, `shape`, `dtype`, +`size`, `fill_value`, `chunking`) and one open-world mapping (`attributes`) for +the CF attributes themselves. `netcdf/_dataset.py` implements them over the +existing `_thread_safe_nc` / `_bytecoding_datasets` wrappers, absorbing the +`ncattrs`/`getncattr`/`setncattr`/`chunking()`/`file_format`/ +`set_auto_chartostring`/`createVariable`/`createDimension` calls that are +scattered across six modules today. `CFVariable` then holds a +`CFDatasetVariable` instead of a `netCDF4.Variable`, resolves `__getattr__` +against a read-tracking view of `.attributes` instead of caching into +`__dict__`, and keeps a one-release deprecating fallback to the backing netCDF4 +object so that `cf_var.getncattr("units")` still works for downstream code. + +**Tech Stack:** Python 3.12–3.14, `netCDF4`, NumPy, Dask, pytest + +pytest-xdist, pre-commit (ruff, numpydoc, mypy, sort-all), `abc.ABC`, +`collections.abc.MutableMapping`, `types.MappingProxyType`. + +**Spec:** [`../specs/2026-09-21-zarr-io-design.md`](../specs/2026-09-21-zarr-io-design.md) +— §4.2 defines the two abstract classes, §4.3 the `__getattr__` rewiring and +the attribute-tracking hazard, §4.4 the `spans` gap this fixes, §4.5 the write +proxy and the `mode`/`finalise` provisions, §5's second entry is this pull +request, §6 the testing rules. The plan argues from the spec; read both. + +--- + +## Global Constraints + +Project-wide requirements every task below implicitly carries. Values are +copied from the spec and the repository's agent rules, not restated from +memory. + +| Constraint | Value | Source | +|---|---|---| +| Target branch | `brownfield`. **Not** `main` | spec §5 | +| Labels | `Agentic` and `Type: Feature Branch` | spec §5 | +| Issue linkage | `Part of #6977` — never a closing keyword; those live on the merge-back | spec §5 | +| Changelog | Fragments land with the merge-back; the pull request body must say the omission is deliberate | spec §5 | +| This pull request | "**No behaviour change other than the `spans` fix. The suite from PR 1 must pass untouched.**" | spec §5, PR 2 | +| `create_variable` keywords | `**encoding` is deliberate, "against the general rule about keyword passthrough, because the accepted keys are backend-specific by nature". Each implementation documents its own accepted keys, and the saver never constructs them | spec §4.2 | +| `__getattr__` | Resolves against `self.attributes`; **no `setattr(self, name, value)` caching**; a missing key raises plain `AttributeError` | spec §4.3 | +| Attribute tracking | "`.attributes` is therefore a tracking mapping, not a plain `dict`" — `cf_attrs_unused()` decides which attributes reach the cube | spec §4.3 | +| netCDF-only coercion | "`_bytes_if_ascii` and `_setncattr` … That coercion stays on the netCDF path only" | spec §4.5 | +| `write_handle()` | "is needed today, by netCDF". `NetCDFDatasetVariable` returns the write proxy — in practice `EncodedNetCDFWriteProxy`; "The base class survives as public API and as the superclass. The relocation must preserve both" | spec §4.5 | +| Not fixed here | The `da.store(..., compute=False)` tuple-return defect, [#7291](https://github.com/SciTools/iris/issues/7291). PR 2 relocates the call site unchanged | spec §4.5 | +| Tests | "pytest style, no network, no unittest classes in new files" | spec §6 | +| Contract test | "The `CFDataset` contract gets one shared test body run against both implementations, so a netCDF/Zarr divergence fails loudly" | spec §6 | +| Every commit | `pre-commit run --files ` then `pytest`, both **before** the commit. Never `--no-verify` | `AGENTS.md` | +| Test tiers | `pytest lib/iris/tests/unit//` per edit (no `-n auto` on small runs); `-n auto` only for the parent-directory and full-suite runs | `lib/iris/tests/AGENTS.md` | +| Test data | `OVERRIDE_TEST_DATA_REPOSITORY` must be set **and verified** before trusting any run; read the skip count, not just the failures | `lib/iris/tests/AGENTS.md` | +| Line length | 88 characters, Ruff default | `AGENTS.md` | +| Docstrings | NumPy style, numpydoc-validated. Module docstring mandatory outside tests (D100) | `AGENTS.md` | +| New `.py` files | Four-line copyright header, checked by `check-license-headers` | `AGENTS.md` | +| netCDF access | No direct `import netCDF4` — always `iris.fileformats.netcdf._thread_safe_nc` | `AGENTS.md` | +| `__getattr__` | Permitted only for "open-world file data keys forwarding to a declared Mapping", and that Mapping must be the path library code takes | `lib/iris/AGENTS.md` | +| No string dispatch | No `getattr(obj, computed_name)` to choose behaviour; the two computed-name sites in this diff become mapping lookups | `lib/iris/AGENTS.md` | +| New modules | Aim under ~1000 lines; cohesion wins over line count | `lib/iris/AGENTS.md` | +| Deprecation | `iris._deprecation.warn_deprecated()`; NEP 29 schedule; mark `.. deprecated:: 3.17` | `AGENTS.md` | +| Attribution | Commits end `Co-Authored-By: Claude Opus 5 `; the pull request body ends with the Claude Code line | repository convention | + +--- + +## Review Focus + +Five failure modes the spec implies that **nothing in the existing suite +exercises**, most likely first. Each names the condition and the behaviour a +reasonable person expects; each has a test added to the task that owns the +code, written out in that task's own step style. + +1. **A file attribute whose name collides with a `CFVariable` class member.** + A variable carrying an attribute called `filename`, `cf_name`, `spans`, + `attributes` or `cf_data` is silently shadowed by the class member of that + name — spec §4.3 documents this as a known limitation rather than fixing it, + because CF reserves no namespace. Nothing today checks that the *mapping* + still yields the file value, which is the whole practical argument for + making `.attributes` the library path. The shadowing must be one-way: + `cf_var.filename` is the file path, `cf_var.attributes["filename"]` is the + file attribute, and `cf_attrs_unused()` still reports it so it reaches the + cube. → **Task 6, Steps 9–10**. + +2. **`getattr(cf_var, name, default)` and `hasattr(cf_var, name)` after the + rewiring.** Spec §4.3 calls these "a deliberate idiom throughout the + loader" and requires a miss to raise "plain `AttributeError`". Two things + can silently break: a `KeyError` or `TypeError` escaping `__getattr__` + (which makes `getattr(..., None)` raise instead of returning the default), + and a change in whether `hasattr` marks the attribute *used* — today it + does, because `hasattr` goes through `__getattr__`. If `hasattr` stops + tracking, every attribute probed but not read leaks onto loaded cubes. + → **Task 6, Steps 11–12**. + +3. **An attribute read again after `cf_attrs_reset()`.** Today `__getattr__` + caches with `setattr(self, name, value)`, and `cf_attrs_reset()` clears the + `_cf_attrs` set but **not** the cached instance attributes **[verified: + `_variables.py:202-210` and `:250`]**. So an attribute touched during + `CFReader._translate` and read again by the loading rules hits `__dict__`, + never re-enters `__getattr__`, and is reported *unused* — landing on the + cube. Removing the cache makes the second read re-register as used, and the + attribute silently stops reaching the cube. This is the single most likely + way this pull request breaks user-visible behaviour, and it is invisible to + a unit test of `CFVariable` alone. → **Task 6, Steps 13–14** (unit) and + **Task 11, Step 1** (`TestAttributeTrackingAcrossReset`, a real file, end + to end). + +4. **`iris.save(cube, )` — the Xarray bridge.** + `Saver.__init__` duck-types its argument with + `hasattr(filename, "createVariable")` (`saver.py:426`) and accepts an + emulating object carrying `_data_array` instead of file-backed storage + (`saver.py:2612`, `loader.py:242`, issue #4994). Routing the saver through + `CFDataset` puts a wrapper between that object and the code that probes it, + and the probe is a `hasattr` on a private name — exactly the kind of thing a + wrapper swallows. A user passing their own dataset must still get their + arrays populated, not an empty file. → **Task 12, Steps 1–3** + (`test_Saver__user_dataset.py`, written green and then proved to bite). + +5. **An attribute value that is not ASCII-encodable, written through the + relocated coercion.** `_bytes_if_ascii` (`saver.py:271`) returns non-ASCII + `str` values *unchanged* and only encodes what round-trips as ASCII; spec + §4.5 says this coercion "stays on the netCDF path only". Moving it behind + `attributes.__setitem__` is where a plausible simplification — "just encode + it" — silently corrupts a UTF-8 `long_name`, or raises where today it + succeeds. → **Task 4, Step 10**. + +--- + +## 1. Scope boundary + +**In scope.** Every access, in `lib/iris/fileformats/cf/`, +`lib/iris/fileformats/_nc_load_rules/` and `lib/iris/fileformats/netcdf/`, +that either + +- reads a CF attribute off a `CFVariable` by attribute syntax + (`getattr(cf_var, ...)`, `hasattr(cf_var, ...)`, `cf_var.units`), or +- calls a netCDF-only API on a variable or dataset (`ncattrs`, `getncattr`, + `setncattr`, `chunking()`, `file_format`, `set_auto_chartostring`, + `createVariable`, `createDimension`, `filepath()`, `isopen()`). + +Spec §2.1 counts "86 `getattr(cf_*` or `hasattr(cf_*` call sites and 57 direct +uses of netCDF-only APIs". Those counts match this tree exactly **[verified]**, +but they are shaped by the `cf_*` naming: `netcdf/ugrid_load.py` reaches CF +attributes through locals named `mesh_var`, `coord_var` and `connectivity_var` +(`ugrid_load.py:230, 322-406`), and those count too. **The working definition +is the bullet list above, not the number.** + +**Out of scope, deliberately.** + +- Moving `netcdf/loader.py` or `netcdf/saver.py`. That is PRs 3 and 5, and + those are specified as pure `git mv`s — which is only achievable because + this pull request makes both modules backend-agnostic first. +- Anything Zarr. No `fileformats/zarr/` directory, no new dependency, no + lock-file change. +- The `da.store(..., compute=False)` tuple-return defect, + [#7291](https://github.com/SciTools/iris/issues/7291). Spec §4.5 is explicit + that this programme does not fix it. +- `_thread_safe_nc.py` and `_bytecoding_datasets.py` keep their current + contents and public names. `netcdf/_dataset.py` is a layer *over* them, not + a replacement. +- The sixty-three grid-mapping parameter assignments in + `_add_grid_mapping_to_dataset` (`saver.py:2067-2272`). They bypass + `_bytes_if_ascii` today, so routing them through `attributes` would change + the CDL Iris writes for every projected file. Task 12 moves them onto a + named netCDF4 handle and leaves the bytes alone. See **finding F8**; Task 13 + records it as an open question in the spec. + +**netCDF-only members that stay netCDF-only.** A handful of things the loader +and saver need have no place on a format-agnostic ABC: whether a variable is a +netCDF VLEN type, whether an emulating object carries `_data_array`, the write +lock. These become **named members on `NetCDFDatasetVariable` / +`NetCDFDataset`**, reached from netCDF modules through `cf_var.cf_data`. They +are named rather than probed, which is the point. Where they live after PRs 3 +and 5 split the loader and saver is those pull requests' decision; this one +only has to make them nameable. + +--- + +## 2. Findings that refine the spec + +Fifteen things the spec does not settle, which the implementation has to. +F1–F6 refine the §4.2 sketch, and none is a discovery of an error: §4.2 calls +its own listing "deliberately small: only what the CF variable classes, +`cf/loader.py` and `cf/saver.py` actually need", and these are what they turn +out to need. F7–F13 are things found by reading the code the pull request +touches, two of which change what spec §5 said this pull request would do. +F14 and F15 are things found by *running* the suite on the baseline commit. +Neither is this pull request's code to fix, and both will waste your time if +you meet them without warning. F15 has since been resolved — it was the +environment, not the repository — and resolving it is what makes the +baseline green and this pull request's negative claim checkable at all. **Task 13 records F1–F6, F8, F9 and F13 in the +spec itself**, so the ABC in §4.2 and the ABC in the code do not drift, and so +the programme's decision log carries the two behaviour changes. + +**F1 — Tracking lives on `CFVariable`, not on `CFDatasetVariable`.** §4.2 +annotates `CFDatasetVariable.attributes` "materialised once; tracks reads", +while §4.3 has `CFVariable.__getattr__` resolve "against `self.attributes`". +They cannot both own the tracking state, because `CFReader` builds a *second* +`CFVariable` over the *same* backing variable when it promotes one +(`_reader.py:505` `CFDataVariable(cf_var_name, cf_var.cf_data)`, and +`:518`), and today those two have independent `_cf_attrs` sets. Tracking on the +dataset variable would make them share one, changing which attributes reach +each cube. **Decision:** `CFDatasetVariable.attributes` is a plain +`MutableMapping`, materialised once; `CFVariable.attributes` is a +`TrackedAttributes` view over it, one per `CFVariable`. Behaviour is then +identical to today's. + +**F2 — `CFDatasetVariable` gains `location: str`.** `CFVariable.filename` +(`_variables.py:107-111`) is `data.group().filepath()` today, and it is +load-bearing twice: it is `NetCDFDataProxy.path`, which appears in +`repr(proxy)` and is therefore part of the dask array cache key +(`_thread_safe_nc.py:368-374`), and it is `LOAD_PROBLEMS.record(filename=...)`. +A variable must be able to answer it without a back-reference to its dataset. +§4.2 already gives `CFDataset.location` exactly this meaning — "path or URL, +for messages and proxies" — so the member is repeated on the variable with the +same meaning and the existing `""` fallback. + +**F3 — `CFDataset` gains `closed: bool`.** `Saver.complete()` guards on +`self._dataset.isopen()` (`saver.py:2694`) and must keep doing so. `isopen()` +is netCDF vocabulary, and the saver has to be backend-agnostic before PR 5 can +`git mv` it, so the state is declared on the ABC as a property. + +**F4 — `create_variable`'s `dimensions` defaults to `()`.** Grid-mapping +variables are scalar: `self._dataset.createVariable(cs.grid_mapping_name, +np.int32)` (`saver.py:2080`) passes no dimensions at all. + +**F5 — `CFDatasetVariable` declares `__len__`.** `len(cf_var)` is part of +`CFVariable`'s surface (`_variables.py:214-215`, pinned by PR 1's +`test_getitem_and_len_delegate_to_underlying_variable`). The ABC gives it a +concrete default of `self.shape[0]`; `NetCDFDatasetVariable` overrides it to +delegate, preserving netCDF4's exact behaviour including on scalars. + +**F6 — `finalise()` is defined but not called.** Spec §4.5 requires it on the +ABC from the start and says "For netCDF, `finalise()` is a no-op". Wiring a +no-op into `Saver.__exit__` would be untestable speculation about an ordering +only Zarr can exercise. PR 6 wires it, in `cf/saver.py`, where it does +something. PR 5's `git mv` is unaffected either way. + +**F7 — "The suite from PR 1 must pass untouched" holds with two exceptions, +both in `lib/iris/tests/unit/fileformats/cf/`.** PR 1's +`test_CFVariable.py::TestAttributeAccess::test_cached` and +`::test_getattr_non_ncattr_value_is_cached_but_not_marked_used` assert the +`setattr(self, name, value)` caching that spec §4.3 requires this pull request +to delete — `assert "coordinates" in cf_var.__dict__` (`test_CFVariable.py:153`) +is the cache, named. They are rewritten in Task 6. + +Separately, the mock-netCDF fixtures in `test_CFVariable.py` and +`identify_mixins.py` construct `CFVariable`s directly from a `MagicMock` +netCDF variable. They change twice, and the two changes are deliberately kept +apart. **Task 6** gives them a working `getncattr`, because `CFVariable` +starts reading attributes through the netCDF4 API rather than off +`ncattrs()`-driven `setattr` caching; the doubles are still netCDF4 doubles. +**Task 11** wraps them in a real `NetCDFDatasetVariable`, or gives them an +`attributes` mapping, because `cf_data` stops being a netCDF4 object at all. +Their assertions are untouched by either. + +**Every other test file in PR 1's suite — the other seventeen modules, plus +`lib/iris/tests/unit/fileformats/nc_load_rules/` and the integration suite — +must pass byte-identically.** Task 13 verifies that by diffing a full-suite run +against the baseline commit. + +**F8 — sixty-three grid-mapping parameters keep a raw netCDF4 handle.** +`_add_grid_mapping_to_dataset` (`saver.py:2067-2272`) writes every CF +grid-mapping parameter by Python attribute assignment — +`cf_var_grid.longitude_of_projection_origin = ...` — not through +`_setncattr`. So they bypass `_bytes_if_ascii`, and `crs_wkt` (`:2272`) lands +as `NC_STRING` where `grid_mapping_name` lands as `NC_CHAR`. Routing them +through `attributes.__setitem__` would apply the coercion and change the CDL +of every projected file Iris writes — a real output change, hiding inside a +refactor that promises none. **Decision:** Task 12 Step 8 binds +`grid_variable = cf_var_grid.variable` once and renames the sixty-three +assignments onto it, leaving what reaches the file identical. Recorded as +open question Q8 in the spec; regularising it is its own pull request. + +**F9 — the `spans` gap spec §4.4 calls a latent bug is unreachable +[verified].** §4.4 says the three `CFVariable` subclass `spans` overrides +lack the `len(dimensions) == 1` guard that `CFVariable.spans` has, so a +variable with `_NCZARR_SCALAR_DIMENSION` *plus* a real dimension would be +wrongly treated as scalar. But `_NCZARR_SCALAR_DIMENSION` is NCZarr's marker +for a variable that has *no* dimensions; it never appears alongside another +one, in any file Iris can read. The guard can therefore never fire, and +"fixing" it would be a change no test can distinguish from a no-op. +**Decision:** Task 7 unifies the four implementations onto one helper — +which is the maintainability half of what §4.4 wanted — and pins the current +behaviour with a characterisation test, explicitly including the +two-dimension case §4.4 imagined. The "one behaviour change" §5 budgets is +spent on F13 instead. + +**F10 — `build_raw_cube` may start showing a synthesised `bounds`.** Raw +loading copies `cf_var.cf_attrs_unused()` onto the cube. `CFReader._build` +sets `bounds` on a promoted variable's `CFVariable` in a couple of places +(`_reader.py`), and with the `setattr` cache gone those writes have to land +somewhere — they land in `attributes`, where `cf_attrs_unused()` can see +them. Task 10 Step 3 therefore routes `CFReader`'s own writes through +`attributes` **and** marks them read, so raw cubes carry exactly what they +carry today. The `test_cf_raw_attrs` integration tests are the check. + +**F11 — `CFVariable.attributes` is a snapshot, not a live view.** +`TrackedAttributes` wraps its source without copying, which is right for a +`dict`. It is wrong over `_NetCDFAttributes`, whose `__setitem__` calls +`setncattr` — a write to the file. `CFReader` assigns to a `CFVariable`'s +attributes during loading (F10), and a read-mode load must never write. +**Decision:** `CFVariable.__init__` does +`TrackedAttributes(dict(data.attributes), ignored=_CF_ATTRS_IGNORED)`. The +copy is a few dozen short strings per variable and is taken once. + +**F12 — `_NetCDFAttributes` must cache the value as given, not as coerced +[verified].** `__setitem__` applies `_bytes_if_ascii` on the way to +`setncattr`, so an ASCII `str` goes to the file as `bytes`. netCDF4 hands it +back as `str`: `setncattr(nm, b"hello")` then `getncattr(nm)` returns +`'hello'`. If the in-memory cache stores the coerced value, a just-written +attribute reads back as `bytes` while the same attribute read from a file +reads back as `str` — and `saver.py` compares exactly that: +`cf_var.attributes["formula_terms"] != formula_terms` (`:1184`) and +`" ".join(coords)` (`:1213`). **Decision:** coerce only on the way out; cache +what the caller gave. Task 4 Step 8 has the two-line body. + +**F13 — attribute names that collide with a netCDF4 Python member now reach +the file [verified].** `_set_cf_var_attributes` guards each write with +`if not hasattr(cf_var, name)` (`saver.py:1802`), meaning "don't clobber a +netCDF4 member". But `setncattr("shape", ...)` is perfectly legal — verified +against netCDF4 for `shape`, `name`, `dimensions`, `size`, `dtype`, +`datatype` and `mask` — so the guard silently *drops* a coordinate attribute +with one of those names instead of protecting anything. Once the write goes +through `attributes`, `hasattr` is the wrong question anyway: a +`NetCDFDatasetVariable` has different members from a `netCDF4.Variable`, so +keeping `hasattr` would change *which* attributes get dropped, which is worse +than either answer. **Decision:** Task 12 Step 9 makes it +`if name not in cf_var.attributes:` — "don't overwrite an attribute already +set", which is what the line is for. This is the saver-side half of the same +defect the `__getattr__` rewrite fixes on the load side, and it is the one +behaviour change this pull request spends its budget on. + +**F14 — `integration/netcdf/test_coord_systems.py`'s module fixture errors +depending on what else is running [verified].** Its `_setup` +(`test_coord_systems.py:171-176`) is + +```python +@pytest.fixture(autouse=True, scope="module") +def _setup(tmp_path_factory): + if not hasattr(tlc, "TMP_DIR"): + tlc.TMP_DIR = tmp_path_factory.mktemp("temp") + yield + delattr(tlc, "TMP_DIR") +``` + +— the `yield` is *inside* the `if`. When `tlc.TMP_DIR` is already set, the +generator returns without yielding and all ten tests in `TestCoordSystem` +error with `ValueError: _setup did not yield a value`. (Corrected in Task 13, +which reproduced it: this is **deterministic on module import order**, not on +worker scheduling. `test_coord_systems.py:18` imports +`unit/fileformats/netcdf/loader/test_load_cubes.py` as `tlc`, and that module +sets `TMP_DIR` as a global at import — so any run that imports the unit module +into the same process first errors all ten, serially, with no xdist at all, +and the reverse order does not.) Under `-n auto` it therefore depends on which +modules a worker is handed, which is why on one and the same commit, minutes +apart, it appears in a `-n auto unit/fileformats/ integration/` run and does +**not** appear in a `-n auto lib/iris/tests` run. It also does **not** appear +when `integration/netcdf/` is run serially [all three verified]. + +**This matters more now than it looks.** Since F15 was fixed the rest of the +suite is clean, so these ten errors are the *entire* difference between the +tier baseline and zero — the one piece of noise left, and the only thing that +could be mistaken for damage you did. **Decision:** out of scope, and named +here so it is recognised rather than debugged. Take the baseline for each +tier you intend to compare (§6.1) and it cancels out; if you would rather not +have to think about it, run that tier serially or drop +`integration/netcdf/test_coord_systems.py` from the selection, and both +sides go to zero. + +**F15 — a stale `iris-test-data` removed exactly the coverage Task 9 needs +[verified, and fixed before this plan was finalised].** The checkout on this +machine was at `1696ac3`, a hundred commits behind +`SciTools/iris-test-data` at `f4a6a05`, and the gap contained almost the +whole `NetCDF/unstructured_grid/` set — **one** file present where there +should be thirteen. All thirteen tests in +`integration/netcdf/test_ugrid_load.py` failed as a result, and +`ugrid_load.py` is what Task 9 rewrites. + +This is the hazard `lib/iris/tests/AGENTS.md` describes as a silent skip, +here wearing a failure's clothes instead of a skip's and therefore *more* +dangerous: 38 red tests look like pre-existing noise you have already agreed +to ignore, so a real regression landing on top of them is invisible. + +**Resolved.** The remote was already configured, so the fix was two commands: + +```bash +git -C ../iris-test-data fetch upstream +git -C ../iris-test-data merge --ff-only upstream/master +# now at f4a6a05; NetCDF/unstructured_grid/ holds 13 files, was 1 +``` + +With that done **the full suite is green — 0 failed, 0 errors** (§6.2), which +is what makes this pull request's negative claim checkable at all. §6.1 keeps +the one-line environment proof as a precondition of Task 1, because a fresh +clone or a different machine can land back in the old state without saying so. + +--- + +## 3. File Structure + +### Created + +| File | Responsibility | Task | +|---|---|---| +| `lib/iris/fileformats/cf/dataset.py` | `CFDataset`, `CFDatasetVariable`, `TrackedAttributes`. Pure interface plus the one concrete mapping type they share. No netCDF import. ~220 lines | 1 | +| `lib/iris/fileformats/netcdf/_dataset.py` | `NetCDFDataset`, `NetCDFDatasetVariable`, `_NetCDFAttributes`. The single place `ncattrs`/`getncattr`/`setncattr`/`chunking()`/`file_format`/`set_auto_chartostring`/`createVariable`/`createDimension` are called. ~400 lines | 2, 3, 4 | +| `lib/iris/tests/unit/fileformats/cf/dataset/__init__.py` | Package marker | 1 | +| `lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py` | Tracking semantics: what records a read and what does not | 1 | +| `lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py` | The ABCs are abstract; the concrete defaults (`__len__`, `deprecated_netcdf_member`) behave | 1 | +| `lib/iris/tests/unit/fileformats/cf/dataset/contract.py` | `class CFDatasetContract` — the shared test body spec §6 requires, parametrised on a `dataset` fixture the backend supplies. Not collected itself (no `Test` prefix) | 5 | +| `lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py` | Package marker | 2 | +| `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py` | Read surface over a real file in `tmp_path` | 2 | +| `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py` | Open / borrow / close, dimensions, global attributes | 3 | +| `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py` | `create_dimension`, `create_variable`, attribute coercion, `write_handle()` | 4 | +| `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py` | `class TestNetCDFDatasetContract(CFDatasetContract)` — supplies the fixture, inherits the body | 5 | +| `lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py` | `CFReader` builds and owns a `NetCDFDataset`; attribute tracking across `cf_attrs_reset()`, end to end on a real file | 11 | +| `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py` | Review Focus 4. Saving into a dataset the caller opened, real and emulated. Written green, and proved to bite | 12 | +| `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py` | `Saver` owns or borrows a `NetCDFDataset`; writes go through it; `complete()` sees a dataset closed behind its back | 12 | + +### Modified + +`lib/iris/fileformats/cf/__init__.py` is **not** touched. Spec §4.1 names +`dataset.py` without a leading underscore, alongside `loader.py` and +`saver.py`, so it is a public module in its own right and the documented import +is `from iris.fileformats.cf.dataset import CFDataset` — the same shape as +§4.8's `iris.fileformats.cf.loader.CHUNK_CONTROL`. The package `__init__.py` +re-export layer covers the three *private* modules and stays exactly as PR 1 +left it, which also leaves PR 1's `test___init__.py` untouched. + +| File | Change | Task | +|---|---|---| +| `lib/iris/fileformats/cf/_variables.py` | `CFVariable.attributes`, the `__getattr__` rewiring, typed properties, the `cf_attrs_*` helpers onto the mapping, the deprecating fallback (Task 6); `_NCZARR_SCALAR_DIMENSION` in three `spans` overrides (Task 7); `cf_data` becomes a `CFDatasetVariable` (Task 11) | 6, 7, 11 | +| `lib/iris/fileformats/_nc_load_rules/helpers.py` | 63 attribute reads onto `.attributes`, including the computed-name site at `:567` and the deliberately-untracked flag probe at `:1213` | 8, 11 | +| `lib/iris/fileformats/_nc_load_rules/actions.py` | 2 attribute reads (`:179`, `:626`) | 8 | +| `lib/iris/fileformats/netcdf/loader.py` | Attribute reads (Task 9); `chunking()`, `cf_data.dimensions`, `_FillValue`, proxy construction (Task 11) | 9, 11 | +| `lib/iris/fileformats/netcdf/ugrid_load.py` | Attribute reads on `mesh_var`/`coord_var`/`connectivity_var`, including the computed-name site at `:406` | 9 | +| `lib/iris/fileformats/cf/_reader.py` | Attribute reads (Task 10); dataset construction, `file_format`, `set_auto_chartostring`, deletion of `_getncattr` (Task 11) | 10, 11 | +| `lib/iris/fileformats/netcdf/saver.py` | `Saver` builds and uses a `NetCDFDataset`; `_bytes_if_ascii`/`_setncattr` relocate; `write_handle()` replaces the direct proxy construction | 12 | +| `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py` | Two caching tests rewritten, Review Focus 1–3 added (Task 6); doubles rewritten (Task 11) | 6, 11 | +| `lib/iris/tests/unit/fileformats/cf/identify_mixins.py` | Stub gains `getncattr` (Task 6); doubles rewritten (Task 11). Finding F7 | 6, 11 | +| `lib/iris/tests/unit/fileformats/cf/test_CFReader.py` | `netcdf_variable()` drives `ncattrs`/`getncattr`; the patch targets move | 11 | +| `lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py` | `_data_variable` helper replaces seven bare `MagicMock`s (Task 9), then changes backing once (Task 11). No assertion changes | 9, 11 | +| `lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py` | `_make`'s `cf_data` becomes a `spec=`'d `NetCDFDatasetVariable`: `chunking` a property, `is_variable_length`, `is_emulated`/`emulated_data_array` | 11 | +| `lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py` | Two `Mock(spec=[])` doubles gain `is_emulated=False`; two `delattr(..., "_data_array")` lines go | 11 | +| `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py` | `test_zlib`'s patch target and expected call; three `createvar_spy` blocks; `_check_bounds_setting` | 12 | +| `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py` | `mock_var` specs `NetCDFDatasetVariable`; `is_emulated` / `emulated_data_array` / `write_handle()` | 12 | +| `docs/superpowers/specs/2026-09-21-zarr-io-design.md` | §4.2 ABC updated for F1–F6; §4.3 says what `CFVariable.attributes` is; §12.1 PR 2 row; §12.3 gains Q8 (F8); §12.4 gains thirteen decisions (five were foreseen here; Tasks 6 and 10-12 added eight more); §12.6 history row | 13 | + +### Sequencing, and why it is this way + +`CFVariable.cf_data` changing type breaks every consumer at once. So the +consumers are rewritten **first**, against a `CFVariable.attributes` that is +initially backed by today's netCDF4 object (Task 6), and the type swap happens +once, late, in Task 11. Every task in between leaves the tree green. + +``` +1 cf/dataset.py (new, unwired) +2 netcdf/_dataset.py variable (new, unwired) +3 netcdf/_dataset.py dataset read (new, unwired) +4 netcdf/_dataset.py dataset write (new, unwired) +5 the shared contract test (new, unwired) + | +6 CFVariable.attributes + __getattr__ rewiring <- still netCDF4-backed + <- behaviour change 1 of 2 +7 one definition of "scalar" across four spans <- characterisation; F9 + | +8 _nc_load_rules/ ---. +9 netcdf/loader.py | consumers, in any order, each green + netcdf/ugrid_load.py | +10 cf/_reader.py ---' + | +11 the swap: CFReader builds a NetCDFDataset; cf_data becomes CFDatasetVariable +12 netcdf/saver.py: Saver builds and uses a NetCDFDataset +13 spec updates, full-suite diff, pull request +``` + +--- + +## 4. The interface, as implemented + +The single reference for every task below. Task 1 writes the first block, +Tasks 2–4 the second. Names here are what later tasks call. + +```python +# lib/iris/fileformats/cf/dataset.py + +class TrackedAttributes(MutableMapping[str, Any]): + def __init__(self, source: MutableMapping[str, Any], + *, ignored: Iterable[str] = ()) -> None: ... + + # Recording lookups. + def __getitem__(self, key: str) -> Any: ... # records `key` as read + def __contains__(self, key: object) -> bool: ... # records a *hit* as read + # (`get` is Mapping's, and therefore records too.) + + # Not recording. + def __setitem__(self, key: str, value: Any) -> None: ... + def __delitem__(self, key: str) -> None: ... + def __iter__(self) -> Iterator[str]: ... + def __len__(self) -> int: ... + def keys(self) -> KeysView[str]: ... + def values(self) -> ValuesView[Any]: ... + def items(self) -> ItemsView[str, Any]: ... + + @property + def untracked(self) -> Mapping[str, Any]: ... # read-only, records nothing + @property + def read(self) -> frozenset[str]: ... + @property + def unread(self) -> frozenset[str]: ... + def reset(self) -> None: ... # back to `ignored` ∩ keys + + +class CFDatasetVariable(ABC): + name: str # abstract property + location: str # abstract property (finding F2) + dimensions: tuple[str, ...] # abstract property + shape: tuple[int, ...] # abstract property + dtype: np.dtype # abstract property + size: int # abstract property + fill_value: Any | None # abstract property + chunking: tuple[int, ...] | None # abstract property; None when unchunked + attributes: MutableMapping[str, Any] # abstract property; materialised once + + def __getitem__(self, keys) -> np.ndarray: ... # abstract + def __setitem__(self, keys, values) -> None: ... # abstract + def write_handle(self) -> Any: ... # abstract + def __len__(self) -> int: ... # concrete: self.shape[0] + def deprecated_netcdf_member(self, name: str) -> Any: ... + # concrete: raises AttributeError + + +class CFDataset(ABC): + location: str # abstract property + mode: str # abstract property; "r"|"r+"|"a"|"w"|"w-" + closed: bool # abstract property (finding F3) + variables: Mapping[str, CFDatasetVariable] # abstract property + dimensions: Mapping[str, int] # abstract property + attributes: MutableMapping[str, Any] # abstract property + + def create_dimension(self, name: str, size: int | None) -> None: ... # abstract + # size=None requests an unlimited dimension: + # saver.py:829 passes it. A store with no such + # concept raises. + def create_variable(self, name: str, dtype, dimensions=(), *, + fill_value=None, **encoding) -> CFDatasetVariable: ... + # abstract + def sync(self) -> None: ... # abstract + def finalise(self) -> None: ... # abstract; no-op for netCDF (finding F6) + def close(self) -> None: ... # abstract + def __enter__(self) -> "CFDataset": ... # concrete + def __exit__(self, *exc_info) -> None: ...# concrete: calls self.close() +``` + +```python +# lib/iris/fileformats/netcdf/_dataset.py + +class _NetCDFAttributes(MutableMapping[str, Any]): + """A netCDF object's attributes: materialised for reads, write-through.""" + def __init__(self, target) -> None: ... # target: VariableWrapper | DatasetWrapper + # __setitem__ applies _bytes_if_ascii coercion, then target.setncattr + + +class NetCDFDatasetVariable(CFDatasetVariable): + def __init__(self, variable, location: str, *, write_lock=None) -> None: ... + # variable: _thread_safe_nc.VariableWrapper | _bytecoding_datasets.EncodedVariable + # write_lock: passed down by NetCDFDataset so that every variable of one + # dataset shares ONE lock. _dask_locks.get_worker_lock() returns a fresh + # threading.Lock under the threaded scheduler, so calling it per variable + # would hand out locks that exclude nothing. + + @property + def variable(self): ... # the backing wrapper, for netCDF-only use + @property + def is_variable_length(self) -> bool: ... # netCDF VLType + @property + def is_emulated(self) -> bool: ... # backed by an object carrying _data_array + @property + def emulated_data_array(self): ... # Xarray bridge, #4994 + @emulated_data_array.setter + def emulated_data_array(self, value) -> None: ... + # is_emulated and emulated_data_array are a pair, because the thing they + # replace is `hasattr(cf_var, "_data_array")` - a PRESENCE test, which a + # None-returning property could not reproduce: an emulator is free to + # initialise _data_array to None and still be emulated. + def deprecated_netcdf_member(self, name: str) -> Any: ... # -> backing wrapper + def write_handle(self): ... # -> EncodedNetCDFWriteProxy + + +class NetCDFDataset(CFDataset): + def __init__(self, location, mode: str = "r", *, netcdf_format=None, + warn_legacy_format: bool = False) -> None: ... + + @classmethod + def from_existing(cls, dataset) -> "NetCDFDataset": ... # borrowed; never closed + + @property + def dataset(self): ... # the backing DatasetWrapper, for netCDF-only use + @property + def write_lock(self): ... # _dask_locks.get_worker_lock(location), one per dataset +``` + +--- + +## 5. Tasks + +### 5.0 The green check every task ends with + +Before **every** commit, in this order. No task below repeats it; it is assumed. + +```bash +# 1. Confirm the test-data path took effect. A missing repository SKIPS tests +# silently, so an unset variable removes coverage without failing anything; +# a *stale* one fails them loudly and looks like noise (F15). +echo "${OVERRIDE_TEST_DATA_REPOSITORY:?set OVERRIDE_TEST_DATA_REPOSITORY first}" +ls "$OVERRIDE_TEST_DATA_REPOSITORY/NetCDF/unstructured_grid/" | wc -l # 13 + +# 2. Hooks rewrite files in place: re-stage and re-run until clean. +pre-commit run --files + +# 3. The tier the task's own steps name. Never --no-verify on the commit. +``` + +Read the **skip count**, not only the failure count. A skip is neither a pass +nor a failure, so a `FAILED|ERROR` diff cannot see coverage that silently +stopped running. §6.2 gives the baseline skip count for every selection the +tasks below use; "green" means matching it, not beating it. + +Two pieces of expected noise, so you recognise them rather than debug them: +ten `test_coord_systems.py` errors whenever `integration/` is run under +`-n auto` (F14), and one skip apiece in `unit/fileformats/cf/` and +`unit/fileformats/netcdf/` (§6.2). Everything else is green at baseline. + +--- + +### Task 1: `cf/dataset.py` — the interface and its tracking mapping + +**Files:** +- Create: `lib/iris/fileformats/cf/dataset.py` +- Create: `lib/iris/tests/unit/fileformats/cf/dataset/__init__.py` +- Test: `lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py` +- Test: `lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py` + +**Interfaces:** +- Consumes: nothing. This is the first task. +- Produces: everything in §4's first code block. Later tasks rely on + `TrackedAttributes(source, *, ignored=())` with `.untracked`, `.read`, + `.unread`, `.reset()`; `CFDatasetVariable` with abstract `name`, `location`, + `dimensions`, `shape`, `dtype`, `size`, `fill_value`, `chunking`, + `attributes`, `__getitem__`, `__setitem__`, `write_handle()` and concrete + `__len__`, `deprecated_netcdf_member(name)`; `CFDataset` with abstract + `location`, `mode`, `closed`, `variables`, `dimensions`, `attributes`, + `create_dimension(name, size)`, + `create_variable(name, dtype, dimensions=(), *, fill_value=None, **encoding)`, + `sync()`, `finalise()`, `close()` and concrete `__enter__`/`__exit__`. + +- [ ] **Step 1: Create the test package marker** + +```bash +mkdir -p lib/iris/tests/unit/fileformats/cf/dataset +cat > lib/iris/tests/unit/fileformats/cf/dataset/__init__.py <<'EOF' +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :mod:`iris.fileformats.cf.dataset`.""" +EOF +``` + +- [ ] **Step 2: Write the failing tests for `TrackedAttributes`** + +Create `lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.cf.dataset.TrackedAttributes`.""" + +import pytest + +from iris.fileformats.cf.dataset import TrackedAttributes + +# The set CFVariable seeds tracking with: attributes netCDF4 handles itself, +# and which are therefore "used" before anyone reads them. +IGNORED = ("_FillValue", "add_offset", "missing_value", "scale_factor") + + +@pytest.fixture +def source(): + return {"units": "K", "standard_name": "air_temperature", "_FillValue": -999} + + +@pytest.fixture +def tracked(source): + return TrackedAttributes(source, ignored=IGNORED) + + +class TestWhatRecordsARead: + def test_starts_with_only_the_present_ignored_names(self, tracked): + # "scale_factor" is ignored but absent, so it is not seeded. + assert tracked.read == frozenset(["_FillValue"]) + assert tracked.unread == frozenset(["units", "standard_name"]) + + def test_getitem_records(self, tracked): + assert tracked["units"] == "K" + assert tracked.read == frozenset(["_FillValue", "units"]) + assert tracked.unread == frozenset(["standard_name"]) + + def test_get_records(self, tracked): + assert tracked.get("units") == "K" + assert "units" in tracked.read + + def test_contains_records_a_hit(self, tracked): + # hasattr(cf_var, name) goes through __getattr__ today and marks the + # attribute used; the mapping form has to do the same. + assert "units" in tracked + assert "units" in tracked.read + + def test_contains_does_not_record_a_miss(self, tracked): + assert "nonesuch" not in tracked + assert tracked.read == frozenset(["_FillValue"]) + + def test_getitem_of_a_missing_key_raises_before_recording(self, tracked): + with pytest.raises(KeyError, match="nonesuch"): + tracked["nonesuch"] + assert tracked.read == frozenset(["_FillValue"]) + + def test_get_of_a_missing_key_records_nothing(self, tracked): + assert tracked.get("nonesuch") is None + assert tracked.read == frozenset(["_FillValue"]) + + +class TestWhatDoesNotRecordARead: + def test_untracked_getitem(self, tracked): + assert tracked.untracked["units"] == "K" + assert tracked.read == frozenset(["_FillValue"]) + + def test_untracked_contains(self, tracked): + # The form helpers.py needs for its deliberately-unmarked flag probe. + assert "units" in tracked.untracked + assert tracked.read == frozenset(["_FillValue"]) + + def test_untracked_is_read_only(self, tracked): + with pytest.raises(TypeError): + tracked.untracked["units"] = "m" + + def test_iteration(self, tracked): + assert sorted(tracked) == ["_FillValue", "standard_name", "units"] + assert tracked.read == frozenset(["_FillValue"]) + + def test_keys_values_items(self, tracked): + assert sorted(tracked.keys()) == ["_FillValue", "standard_name", "units"] + assert sorted(tracked.values(), key=str) == [-999, "K", "air_temperature"] + assert dict(tracked.items())["units"] == "K" + assert tracked.read == frozenset(["_FillValue"]) + + def test_len(self, tracked): + assert len(tracked) == 3 + assert tracked.read == frozenset(["_FillValue"]) + + +class TestMutation: + def test_setitem_writes_through_and_does_not_record(self, tracked, source): + tracked["comment"] = "hello" + assert source["comment"] == "hello" + assert tracked.read == frozenset(["_FillValue"]) + + def test_delitem_writes_through_and_forgets_the_read(self, tracked, source): + _ = tracked["units"] + del tracked["units"] + assert "units" not in source + assert tracked.read == frozenset(["_FillValue"]) + + +class TestReset: + def test_reset_returns_to_the_present_ignored_names(self, tracked): + _ = tracked["units"] + _ = tracked["standard_name"] + assert tracked.read == frozenset(["_FillValue", "units", "standard_name"]) + tracked.reset() + assert tracked.read == frozenset(["_FillValue"]) + + def test_reset_sees_attributes_added_since_construction(self, tracked): + tracked["scale_factor"] = 2.0 + tracked.reset() + assert tracked.read == frozenset(["_FillValue", "scale_factor"]) + + +class TestEmptySource: + def test_a_variable_with_no_attributes_is_not_an_error(self): + tracked = TrackedAttributes({}, ignored=IGNORED) + assert tracked.read == frozenset() + assert tracked.unread == frozenset() + assert len(tracked) == 0 + assert "units" not in tracked +``` + +- [ ] **Step 3: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py` +Expected: collection error — +`ModuleNotFoundError: No module named 'iris.fileformats.cf.dataset'` + +- [ ] **Step 4: Write `TrackedAttributes`** + +Create `lib/iris/fileformats/cf/dataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""The format-agnostic description of a CF-conforming array store. + +:class:`CFDataset` and :class:`CFDatasetVariable` are what the rest of the CF +layer is written against. They are deliberately small: only what the +:class:`~iris.fileformats.cf.CFVariable` classes, the CF loader and the CF +saver actually need, with one implementation per storage format +(:mod:`iris.fileformats.netcdf._dataset` today, Zarr next). + +The split that matters is ``attributes`` versus everything else. A netCDF +variable presents ``units`` - a CF attribute read from the file - and +``dimensions`` - a property of the storage - through the same attribute syntax, +and a caller cannot tell which is which. Here, CF attributes are only ever +reached through ``attributes``, and storage properties are named, typed +members. + +``attributes`` is an open-ended set of data keys read from a file, with no +schema to enumerate, which is why :class:`TrackedAttributes` exists and why +:meth:`~iris.fileformats.cf.CFVariable.__getattr__` is allowed to forward to +it. Tracking is the load-bearing part: Iris decides which file attributes +survive onto a loaded cube by asking which ones the loading rules did *not* +read, so a read that goes unrecorded silently leaks a CF-reserved attribute +onto the cube. + +See sections 4.2, 4.3 and 4.5 of +``docs/superpowers/specs/2026-09-21-zarr-io-design.md``. + +""" + +from abc import ABC, abstractmethod +from collections.abc import ( + ItemsView, + Iterable, + Iterator, + KeysView, + Mapping, + MutableMapping, + ValuesView, +) +from types import MappingProxyType +from typing import Any + +import numpy as np + + +class TrackedAttributes(MutableMapping): + """A variable's attributes, recording which of them have been looked up. + + Single-key lookups record: :meth:`__getitem__`, :meth:`get` and + :meth:`__contains__` - the last because ``hasattr(cf_var, name)`` marks an + attribute used today, by reaching ``__getattr__``. Bulk access does not: + iteration, :meth:`keys`, :meth:`values`, :meth:`items` and :func:`len` + leave the record alone, and :attr:`untracked` is the explicit escape hatch + for a single-key lookup that must not count. + + """ + + def __init__(self, source: MutableMapping, *, ignored: Iterable[str] = ()): + """Wrap ``source``, treating any ``ignored`` name it holds as already read.""" + self._source = source + self._ignored = frozenset(ignored) + self._read: set[str] = set() + self.reset() + + def __getitem__(self, key: str) -> Any: + """Return ``key``'s value, recording it as read.""" + # Index first: a KeyError must escape without recording anything. + value = self._source[key] + self._read.add(key) + return value + + def __contains__(self, key: object) -> bool: + """Return whether ``key`` is present, recording a hit as read.""" + result = key in self._source + if result: + self._read.add(key) # type: ignore[arg-type] + return result + + def __setitem__(self, key: str, value: Any) -> None: + """Set ``key``'s value, recording nothing.""" + self._source[key] = value + + def __delitem__(self, key: str) -> None: + """Remove ``key``, and any record of it having been read.""" + del self._source[key] + self._read.discard(key) + + def __iter__(self) -> Iterator[str]: + """Return an iterator over the attribute names, recording nothing.""" + return iter(self._source) + + def __len__(self) -> int: + """Return the number of attributes, recording nothing.""" + return len(self._source) + + def keys(self) -> KeysView: + """Return a view of the attribute names, recording nothing.""" + return self._source.keys() + + def values(self) -> ValuesView: + """Return a view of the attribute values, recording nothing.""" + return self._source.values() + + def items(self) -> ItemsView: + """Return a view of the attribute pairs, recording nothing.""" + return self._source.items() + + def __repr__(self) -> str: + """Return a string representation, recording nothing.""" + return f"{self.__class__.__name__}({dict(self._source)!r})" + + @property + def untracked(self) -> Mapping: + """A read-only view of the same attributes that records nothing.""" + return MappingProxyType(dict(self._source)) + + @property + def read(self) -> frozenset: + """The names looked up since construction or the last :meth:`reset`.""" + return frozenset(self._read) + + @property + def unread(self) -> frozenset: + """The names present but not looked up since the last :meth:`reset`.""" + return frozenset(self._source) - self._read + + def reset(self) -> None: + """Forget every recorded lookup, re-seeding from the ignored names.""" + self._read = set(self._ignored) & set(self._source) +``` + +Note `untracked` returns a proxy over a **copy**: `MappingProxyType` needs a +real mapping, and the backing `_NetCDFAttributes` of Task 2 is not a `dict`. +Callers use it for point lookups, never to observe later mutation. + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py` +Expected: PASS — every test in the file, 0 failed, 0 skipped. + +- [ ] **Step 6: Write the failing tests for the two abstract classes** + +Create `lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.cf.dataset.CFDataset` and friends.""" + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable + + +class MinimalVariable(CFDatasetVariable): + """The smallest thing that satisfies CFDatasetVariable.""" + + name = "air_temperature" + location = "" + dimensions = ("time", "lat") + shape = (3, 4) + dtype = np.dtype("f4") + size = 12 + fill_value = None + chunking = None + attributes: dict = {} + + def __getitem__(self, keys): + return np.zeros(self.shape, dtype=self.dtype)[keys] + + def __setitem__(self, keys, values): + raise NotImplementedError + + def write_handle(self): + return self + + +class MinimalDataset(CFDataset): + """The smallest thing that satisfies CFDataset.""" + + location = "" + mode = "r" + closed = False + variables: dict = {} + dimensions: dict = {} + attributes: dict = {} + + def __init__(self): + self.closes = 0 + + def create_dimension(self, name, size): + raise NotImplementedError + + def create_variable(self, name, dtype, dimensions=(), *, fill_value=None, **kw): + raise NotImplementedError + + def sync(self): + pass + + def finalise(self): + pass + + def close(self): + self.closes += 1 + + +class TestAbstractness: + def test_variable_cannot_be_instantiated(self): + with pytest.raises(TypeError, match="abstract"): + CFDatasetVariable() + + def test_dataset_cannot_be_instantiated(self): + with pytest.raises(TypeError, match="abstract"): + CFDataset() + + @pytest.mark.parametrize( + "name", + [ + "name", + "location", + "dimensions", + "shape", + "dtype", + "size", + "fill_value", + "chunking", + "attributes", + "ndim", + "__getitem__", + "__setitem__", + "write_handle", + ], + ) + def test_variable_declares_member(self, name): + assert name in CFDatasetVariable.__abstractmethods__ or hasattr( + CFDatasetVariable, name + ) + + @pytest.mark.parametrize( + "name", + [ + "location", + "mode", + "closed", + "variables", + "dimensions", + "attributes", + "create_dimension", + "create_variable", + "sync", + "finalise", + "close", + ], + ) + def test_dataset_declares_member(self, name): + assert name in CFDataset.__abstractmethods__ or hasattr(CFDataset, name) + + +class TestConcreteDefaults: + def test_ndim(self): + assert MinimalVariable().ndim == 2 + + def test_len_is_the_leading_dimension(self): + assert len(MinimalVariable()) == 3 + + def test_len_of_a_scalar_matches_netcdf4(self): + # netCDF4.Variable raises TypeError, and CFVariable.__len__ forwards + # to it today, so anything catching that keeps working. + class Scalar(MinimalVariable): + shape = () + + with pytest.raises(TypeError, match="unsized"): + len(Scalar()) + + def test_deprecated_netcdf_member_raises_attribute_error(self): + # A store with no backing netCDF4 object - Zarr, say - has no fallback + # to offer, so the compatibility route simply does not apply. + with pytest.raises(AttributeError, match="getncattr"): + MinimalVariable().deprecated_netcdf_member("getncattr") + + def test_context_manager_closes(self): + dataset = MinimalDataset() + with dataset as entered: + assert entered is dataset + assert dataset.closes == 0 + assert dataset.closes == 1 + + def test_context_manager_closes_on_exception(self): + dataset = MinimalDataset() + with pytest.raises(ValueError, match="boom"): + with dataset: + raise ValueError("boom") + assert dataset.closes == 1 +``` + +- [ ] **Step 7: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py` +Expected: collection error — +`ImportError: cannot import name 'CFDataset' from 'iris.fileformats.cf.dataset'` + +- [ ] **Step 8: Write the two abstract classes** + +Append to `lib/iris/fileformats/cf/dataset.py`: + +```python +class CFDatasetVariable(ABC): + """One named array in a CF-conforming dataset.""" + + @property + @abstractmethod + def name(self) -> str: + """The variable's name within its dataset.""" + + @property + @abstractmethod + def location(self) -> str: + """The path or URL of the dataset holding this variable. + + Repeated from :attr:`CFDataset.location` with the same meaning, so that + a variable can label a message or a data proxy without a reference back + to its dataset. + + """ + + @property + @abstractmethod + def dimensions(self) -> tuple: + """The names of the dimensions this variable spans, in order.""" + + @property + @abstractmethod + def shape(self) -> tuple: + """The variable's shape.""" + + @property + @abstractmethod + def dtype(self) -> np.dtype: + """The variable's stored data type. + + Usually a :class:`numpy.dtype`. netCDF's variable-length string type + reports the builtin :class:`str` instead, and + :func:`iris.fileformats.netcdf.loader._get_cf_var_data` branches on + that, so it is passed through rather than normalised. + + """ + + @property + @abstractmethod + def size(self) -> int: + """The total number of elements in the variable.""" + + @property + @abstractmethod + def fill_value(self) -> Any: + """The store's own fill value, or ``None`` if it has none.""" + + @property + @abstractmethod + def chunking(self) -> tuple | None: + """The variable's storage chunk shape, or ``None`` when unchunked.""" + + @property + @abstractmethod + def attributes(self) -> MutableMapping: + """The variable's CF attributes, materialised once at construction.""" + + @abstractmethod + def __getitem__(self, keys) -> np.ndarray: + """Return the indexed portion of the variable's data.""" + + @abstractmethod + def __setitem__(self, keys, values) -> None: + """Write ``values`` into the indexed portion of the variable.""" + + @abstractmethod + def write_handle(self) -> Any: + """Return a picklable object supporting ``__setitem__`` on this variable. + + This is what a Dask worker receives as a ``da.store`` target, so it must + survive pickling and must remain usable after the dataset it came from + has been closed. + + """ + + @property + def ndim(self) -> int: + """The number of dimensions the variable spans.""" + return len(self.shape) + + def __len__(self) -> int: + """Return the length of the variable's leading dimension.""" + if not self.shape: + # netCDF4.Variable.__len__ raises exactly this for a scalar, and + # CFVariable.__len__ forwards to it today. + msg = "len() of unsized object" + raise TypeError(msg) + return self.shape[0] + + def deprecated_netcdf_member(self, name: str) -> Any: + """Return a netCDF4-only member of the object backing this variable. + + The one-release compatibility route for code that reached netCDF4 API + through a :class:`~iris.fileformats.cf.CFVariable`. A store with no + backing netCDF4 object has nothing to offer and raises, which is the + correct answer rather than a special case. + + """ + raise AttributeError(name) + + +class CFDataset(ABC): + """A CF-conforming array store, open for reading or writing.""" + + @property + @abstractmethod + def location(self) -> str: + """The store's path or URL, for messages and data proxies.""" + + @property + @abstractmethod + def mode(self) -> str: + """The mode the store was opened in: ``r``, ``r+``, ``a``, ``w`` or ``w-``.""" + + @property + @abstractmethod + def closed(self) -> bool: + """Whether :meth:`close` has already released the store.""" + + @property + @abstractmethod + def variables(self) -> Mapping: + """The store's variables, by name.""" + + @property + @abstractmethod + def dimensions(self) -> Mapping: + """The store's dimension lengths, by name.""" + + @property + @abstractmethod + def attributes(self) -> MutableMapping: + """The store's global attributes.""" + + @abstractmethod + def create_dimension(self, name: str, size: int | None) -> None: + """Declare a dimension of the given length. + + ``size=None`` requests an unlimited dimension. A store with no such + concept raises :class:`NotImplementedError`. + + """ + + @abstractmethod + def create_variable( + self, name: str, dtype, dimensions=(), *, fill_value=None, **encoding + ) -> "CFDatasetVariable": + """Create and return a new variable. + + ``**encoding`` is storage-specific by nature - ``zlib``, ``complevel`` + and ``chunksizes`` for netCDF; ``compressors``, ``chunks`` and + ``shards`` for Zarr - so each implementation documents the keys it + accepts, and generic CF code never constructs them. + + """ + + @abstractmethod + def sync(self) -> None: + """Flush buffered writes to the store.""" + + @abstractmethod + def finalise(self) -> None: + """Perform the store's one-shot completion step, after every write. + + Deliberately **not** part of :meth:`close`: a worker that writes one + slab of a store must close its handle without performing a completion + step that only one process may perform. + + """ + + @abstractmethod + def close(self) -> None: + """Release the store's resources.""" + + def __enter__(self) -> "CFDataset": + """Return this dataset, for use as a context manager.""" + return self + + def __exit__(self, *exc_info) -> None: + """Close this dataset on leaving the context, whatever happened in it.""" + self.close() +``` + +- [ ] **Step 9: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/dataset/` +Expected: PASS — every test in both files, 0 failed, 0 skipped. + +- [ ] **Step 10: Confirm the new module is importable standalone and adds no coupling** + +```bash +python -c " +import subprocess, sys +src = open('lib/iris/fileformats/cf/dataset.py').read() +assert 'netcdf' not in src and 'netCDF4' not in src, 'cf/dataset.py must not know netCDF' +print('ok') +" +python -c "from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable, TrackedAttributes; print('ok')" +``` +Expected: `ok` twice. + +- [ ] **Step 11: Green check and commit** + +```bash +pre-commit run --files lib/iris/fileformats/cf/dataset.py \ + lib/iris/tests/unit/fileformats/cf/dataset/__init__.py \ + lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py \ + lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py +pytest lib/iris/tests/unit/fileformats/cf/ +git add lib/iris/fileformats/cf/dataset.py lib/iris/tests/unit/fileformats/cf/dataset/ +git commit -m "$(cat <<'EOF' +Add the CFDataset interface and its tracking attributes mapping + +cf/dataset.py declares what the CF layer needs from a storage format: +typed properties for the storage, and one open-world mapping for the CF +attributes. TrackedAttributes records which attributes were looked up, +which is how Iris decides what survives onto a loaded cube. + +Nothing is wired to it yet. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 2: `netcdf/_dataset.py` — the variable, read surface + +**Files:** +- Create: `lib/iris/fileformats/netcdf/_dataset.py` +- Create: `lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py` +- Create: `lib/iris/tests/unit/fileformats/netcdf/dataset/conftest.py` +- Test: `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py` + +**Interfaces:** +- Consumes: `CFDatasetVariable` from Task 1 — every abstract member listed + there, plus the concrete `__len__` and `deprecated_netcdf_member`. +- Produces: + - `_NetCDFAttributes(target)` — a `MutableMapping`; `target` is anything with + `ncattrs()`, `getncattr()`, `setncattr()` and `delncattr()`, i.e. a + `VariableWrapper`, an `EncodedVariable` or a `DatasetWrapper`. Task 3 uses + it for global attributes; Task 4 adds the write-side coercion. + - `NetCDFDatasetVariable(variable, location, *, write_lock=None)` with, on + top of the ABC, `variable`, `is_variable_length`, `is_emulated`, + `emulated_data_array` (get/set). Tasks 11 and 12 consume all of these. + - the pytest fixtures `sample_path`, and the module constants + `SAMPLE_DIMENSIONS` / `SAMPLE_GLOBALS`, used by Tasks 3, 4 and 5. + +- [ ] **Step 1: Create the test package and its sample file** + +`lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :mod:`iris.fileformats.netcdf._dataset`.""" +``` + +`lib/iris/tests/unit/fileformats/netcdf/dataset/conftest.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""A real netCDF4 file to test the netCDF CFDataset implementation against. + +Real, rather than mocked, because the members being implemented are precisely +the ones whose netCDF4 behaviour is easy to misremember - what ``chunking()`` +returns for a contiguous variable, what ``dimensions`` is for char data, what +``ncattrs()`` includes once ``fill_value=`` has been passed. +""" + +import numpy as np +import pytest + +from iris.fileformats.netcdf import _thread_safe_nc + +#: How the sample file's dimensions are created. None means unlimited. +SAMPLE_DIMENSION_SIZES = {"time": 3, "lat": 4, "nchars": 8, "record": None} + +#: How they should then read back. An unlimited dimension reports the number +#: of records actually written, which here is none. +SAMPLE_DIMENSIONS = {"time": 3, "lat": 4, "nchars": 8, "record": 0} + +#: The sample file's global attributes. +SAMPLE_GLOBALS = {"Conventions": "CF-1.7", "title": "sample"} + +#: The sample file's air_temperature payload. +SAMPLE_AIR = np.arange(12, dtype="f4").reshape(3, 4) + +#: The sample file's label payload, as the strings it encodes. +SAMPLE_LABELS = ["alpha", "beta", "gamma"] + + +@pytest.fixture(scope="session") +def sample_path(tmp_path_factory): + """Write, once per session, a small netCDF4 file and return its path.""" + path = tmp_path_factory.mktemp("cf_dataset") / "sample.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + try: + for name, size in SAMPLE_DIMENSION_SIZES.items(): + dataset.createDimension(name, size) + for name, value in SAMPLE_GLOBALS.items(): + dataset.setncattr(name, value) + + air = dataset.createVariable( + "air_temperature", + "f4", + ("time", "lat"), + fill_value=-999.0, + zlib=True, + chunksizes=(1, 4), + ) + air.setncattr("units", "K") + air.setncattr("standard_name", "air_temperature") + air.setncattr("coordinates", "height") + air[:] = SAMPLE_AIR + + # Scalar, and deliberately unchunked: chunking() answers differently. + height = dataset.createVariable("height", "f4", ()) + height.setncattr("units", "m") + height[:] = 1.5 + + # Char data: the one case where EncodedVariable changes shape, dtype + # and dimensions out from under the wrapper. + label = dataset.createVariable("label", "S1", ("time", "nchars")) + label[:] = np.array( + [list(f"{text:<8}") for text in SAMPLE_LABELS], dtype="S1" + ) + finally: + dataset.close() + + return path + + +@pytest.fixture +def sample_location(sample_path): + """The sample file's path, as the string a CFDataset reports.""" + return str(sample_path) +``` + +- [ ] **Step 2: Write the failing tests for the variable's read surface** + +`lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.netcdf._dataset.NetCDFDatasetVariable`.""" + +from collections.abc import MutableMapping + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDatasetVariable +from iris.fileformats.netcdf import _bytecoding_datasets, _dataset, _thread_safe_nc + +from .conftest import SAMPLE_AIR, SAMPLE_LABELS + + +@pytest.fixture +def raw(sample_path): + dataset = _thread_safe_nc.DatasetWrapper(sample_path, mode="r") + yield dataset + dataset.close() + + +@pytest.fixture +def encoded(sample_path): + dataset = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + yield dataset + dataset.close() + + +def wrap(dataset, name, location, write_lock=None): + return _dataset.NetCDFDatasetVariable( + dataset.variables[name], location, write_lock=write_lock + ) + + +@pytest.fixture +def air(raw, sample_location): + return wrap(raw, "air_temperature", sample_location) + + +@pytest.fixture +def height(raw, sample_location): + return wrap(raw, "height", sample_location) + + +class TestStorageProperties: + def test_is_a_cf_dataset_variable(self, air): + assert isinstance(air, CFDatasetVariable) + + def test_name(self, air): + assert air.name == "air_temperature" + + def test_location(self, air, sample_location): + assert air.location == sample_location + + def test_dimensions_are_a_tuple(self, air): + # netCDF4 answers with a tuple already, but the contract says tuple + # and CFVariable.spans does set arithmetic on it, so pin it. + assert air.dimensions == ("time", "lat") + assert isinstance(air.dimensions, tuple) + + def test_shape_dtype_size_ndim(self, air): + assert air.shape == (3, 4) + assert air.dtype == np.dtype("f4") + assert air.size == 12 + assert air.ndim == 2 + + def test_len(self, air): + assert len(air) == 3 + + def test_scalar_has_no_len(self, height): + assert height.shape == () + with pytest.raises(TypeError, match="unsized"): + len(height) + + def test_fill_value(self, air): + assert air.fill_value == -999.0 + + def test_no_fill_value_is_none(self, height): + assert height.fill_value is None + + def test_chunking_is_a_tuple(self, air): + # netCDF4 answers with a list; the contract says a shape. + assert air.chunking == (1, 4) + + def test_contiguous_chunking_is_none(self, height): + # netCDF4 answers "contiguous" here, and loader.py already treats + # that and None identically, so both collapse to None. + assert height.chunking is None + + +class TestData: + def test_getitem(self, air): + np.testing.assert_array_equal(air[:], SAMPLE_AIR) + + def test_getitem_indexed(self, air): + np.testing.assert_array_equal(air[0], SAMPLE_AIR[0]) + + +class TestAttributes: + def test_is_a_plain_mutable_mapping(self, air): + # Tracking belongs to CFVariable, not here: a CFDatasetVariable can + # have two CFVariables promoted over it - see finding F1. + assert isinstance(air.attributes, MutableMapping) + assert not hasattr(air.attributes, "read") + + def test_contents(self, air): + assert dict(air.attributes) == { + "_FillValue": -999.0, + "units": "K", + "standard_name": "air_temperature", + "coordinates": "height", + } + + def test_is_built_once(self, air): + assert air.attributes is air.attributes + + def test_missing_key_raises_key_error(self, air): + with pytest.raises(KeyError, match="nonesuch"): + air.attributes["nonesuch"] + + def test_unreadable_attribute_becomes_empty_string(self, mocker, raw, sample_location): + # ncattrs() can list a name that getncattr then refuses. _getncattr in + # cf/_reader.py tolerated exactly this with a "" default, and that + # tolerance has to survive the move. + variable = raw.variables["height"] + mocker.patch.object(variable, "ncattrs", return_value=["units", "broken"]) + mocker.patch.object( + variable, + "getncattr", + side_effect=lambda name: ( + "m" if name == "units" else _raise(AttributeError(name)) + ), + ) + wrapped = _dataset.NetCDFDatasetVariable(variable, sample_location) + assert wrapped.attributes["broken"] == "" + + +def _raise(exception): + raise exception + + +class TestNetCDFOnlyMembers: + def test_variable_exposes_the_backing_wrapper(self, air, raw): + assert air.variable is raw.variables["air_temperature"] + + def test_not_variable_length(self, air): + assert air.is_variable_length is False + + def test_not_emulated(self, air): + assert air.is_emulated is False + + def test_emulated_data_array_raises_when_not_emulated(self, air): + with pytest.raises(AttributeError, match="_data_array"): + air.emulated_data_array + + def test_emulated_round_trip(self, air): + # The Xarray bridge hook, issue #4994: an emulating variable carries + # its own array instead of file storage. + air.variable._data_array = np.zeros(3) + assert air.is_emulated is True + np.testing.assert_array_equal(air.emulated_data_array, np.zeros(3)) + air.emulated_data_array = np.ones(3) + np.testing.assert_array_equal(air.variable._data_array, np.ones(3)) + + def test_deprecated_netcdf_member_reaches_the_wrapper(self, air): + assert sorted(air.deprecated_netcdf_member("ncattrs")()) == [ + "_FillValue", + "coordinates", + "standard_name", + "units", + ] + + def test_deprecated_netcdf_member_of_a_missing_name_raises(self, air): + with pytest.raises(AttributeError): + air.deprecated_netcdf_member("no_such_netcdf_member") + + +class TestCharacterData: + def test_raw_wrapper_sees_the_char_dimension(self, raw, sample_location): + label = wrap(raw, "label", sample_location) + assert label.dimensions == ("time", "nchars") + assert label.shape == (3, 8) + assert label.dtype == np.dtype("S1") + + def test_encoded_wrapper_sees_strings(self, encoded, sample_location): + # EncodedVariable drops the trailing char dimension and reports a + # string dtype. The CFDatasetVariable must pass that through, not + # reach around it to the contained netCDF4 variable. + label = wrap(encoded, "label", sample_location) + assert label.dimensions == ("time",) + assert label.shape == (3,) + assert label.dtype == np.dtype("U8") + assert [text.strip() for text in label[:]] == SAMPLE_LABELS +``` + +- [ ] **Step 3: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/` +Expected: collection error — +`ModuleNotFoundError: No module named 'iris.fileformats.netcdf._dataset'` + +- [ ] **Step 4: Write the module and the variable's read surface** + +Create `lib/iris/fileformats/netcdf/_dataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""The netCDF implementation of the :mod:`iris.fileformats.cf.dataset` interface. + +This is the only module that knows both the CF interface and the netCDF4 API. +Everything netCDF-specific that the CF layer used to reach through a +:class:`~iris.fileformats.cf.CFVariable` - ``ncattrs``, ``getncattr``, +``setncattr``, ``chunking()``, ``file_format``, ``createVariable``, +``createDimension``, ``filepath()``, ``isopen()`` - either has a named member +on the interface or lives here as a netCDF-only member. + +See section 4.5 of ``docs/superpowers/specs/2026-09-21-zarr-io-design.md``. + +""" + +from collections.abc import Iterator, MutableMapping +from typing import Any + +import numpy as np + +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable + +from . import _bytecoding_datasets, _thread_safe_nc + +#: What ``netCDF4.Variable.chunking()`` answers for an unchunked variable. +#: ``None`` means the same thing, and arrives from non-version-4 files. +_CONTIGUOUS = "contiguous" + +#: The member an emulating variable carries instead of file storage. See +#: https://github.com/SciTools/iris/issues/4994 "Xarray bridge". +_EMULATED_DATA_ARRAY = "_data_array" + + +class _NetCDFAttributes(MutableMapping): + """A netCDF object's attributes as a mapping: read once, written through. + + Values are read once, at construction, because the interface promises a + mapping whose ``keys`` and ``items`` cost nothing - attribute tracking asks + "which of these went unread?" on every variable of every loaded file, and + a lazily-fetching mapping would make that question expensive. + + """ + + def __init__(self, target): + """Materialise the attributes of ``target``, a netCDF variable or dataset.""" + self._target = target + self._values: dict[str, Any] = {} + for name in target.ncattrs(): + try: + value = target.getncattr(name) + except AttributeError: + # ncattrs() can list a name that getncattr then refuses. The + # netCDF4 library does this for some malformed files, and + # cf/_reader.py's _getncattr tolerated it with this default. + value = "" + self._values[name] = value + + def __getitem__(self, key: str) -> Any: + """Return ``key``'s value.""" + return self._values[key] + + def __setitem__(self, key: str, value: Any) -> None: + """Set ``key``'s value, here and in the netCDF object.""" + self._target.setncattr(key, value) + self._values[key] = value + + def __delitem__(self, key: str) -> None: + """Remove ``key``, here and from the netCDF object.""" + self._target.delncattr(key) + del self._values[key] + + def __iter__(self) -> Iterator[str]: + """Return an iterator over the attribute names, in file order.""" + return iter(self._values) + + def __len__(self) -> int: + """Return the number of attributes.""" + return len(self._values) + + def __repr__(self) -> str: + """Return a string representation.""" + return f"{self.__class__.__name__}({self._values!r})" + + +class NetCDFDatasetVariable(CFDatasetVariable): + """One variable of a netCDF file, presented through the CF interface.""" + + def __init__(self, variable, location: str, *, write_lock=None): + """Wrap ``variable``, a thread-safe netCDF variable wrapper. + + Parameters + ---------- + variable : :class:`~iris.fileformats.netcdf._thread_safe_nc.VariableWrapper` + The wrapped netCDF variable. May be an + :class:`~iris.fileformats.netcdf._bytecoding_datasets.EncodedVariable`, + whose shape, dimensions and dtype differ from the file's own for + character data; those are passed through, not reached around. + location : str + The path or URL of the dataset this variable belongs to. + write_lock : optional + The lock shared by every variable of one dataset, used by + :meth:`write_handle`. Supplied by :class:`NetCDFDataset`. + + """ + self._variable = variable + self._location = location + self._write_lock = write_lock + self._attributes = _NetCDFAttributes(variable) + + @property + def name(self) -> str: + """The variable's name within its dataset.""" + return self._variable.name + + @property + def location(self) -> str: + """The path or URL of the dataset holding this variable.""" + return self._location + + @property + def dimensions(self) -> tuple: + """The names of the dimensions this variable spans, in order.""" + return tuple(self._variable.dimensions) + + @property + def shape(self) -> tuple: + """The variable's shape.""" + return tuple(self._variable.shape) + + @property + def dtype(self): + """The variable's stored data type, or ``str`` for a VLEN string type.""" + return self._variable.dtype + + @property + def size(self) -> int: + """The total number of elements in the variable.""" + return self._variable.size + + @property + def fill_value(self) -> Any: + """The variable's ``_FillValue`` attribute, or ``None`` if it has none.""" + return self._attributes.get("_FillValue") + + @property + def chunking(self) -> tuple | None: + """The variable's storage chunk shape, or ``None`` when unchunked. + + netCDF answers ``None`` for a non-version-4 file and the string + ``"contiguous"`` for an unchunked version-4 variable. Both mean the + same thing to a caller choosing a Dask chunking, so both become + ``None``. + + """ + chunks = self._variable.chunking() + if chunks is None or chunks == _CONTIGUOUS: + return None + return tuple(chunks) + + @property + def attributes(self) -> MutableMapping: + """The variable's CF attributes.""" + return self._attributes + + def __getitem__(self, keys) -> np.ndarray: + """Return the indexed portion of the variable's data.""" + return self._variable[keys] + + def __setitem__(self, keys, values) -> None: + """Write ``values`` into the indexed portion of the variable.""" + self._variable[keys] = values + + def __repr__(self) -> str: + """Return a string representation.""" + return f"{self.__class__.__name__}({self.name!r}, {self.location!r})" + + # netCDF-only members below: named here rather than reached for through + # getattr, so that a Zarr caller fails to import them rather than failing + # at run time with an AttributeError from somewhere unrelated. + + @property + def variable(self): + """The backing thread-safe netCDF variable wrapper.""" + return self._variable + + @property + def is_variable_length(self) -> bool: + """Whether this is a netCDF variable-length (VLEN) type. + + Such a variable's total size cannot be known without reading it - see + https://github.com/Unidata/netcdf-c/issues/1893 - so the loader has to + guess whether it is worth making lazy. + + """ + datatype = getattr(self._variable, "datatype", None) + return isinstance(datatype, _thread_safe_nc.VLType) + + @property + def is_emulated(self) -> bool: + """Whether an emulating object supplies this variable's data directly. + + The Xarray bridge, https://github.com/SciTools/iris/issues/4994: the + "file" is an emulator and its variables carry their own arrays. + + """ + return hasattr(self._variable, _EMULATED_DATA_ARRAY) + + @property + def emulated_data_array(self): + """The array an emulating variable carries instead of file storage.""" + return getattr(self._variable, _EMULATED_DATA_ARRAY) + + @emulated_data_array.setter + def emulated_data_array(self, value) -> None: + setattr(self._variable, _EMULATED_DATA_ARRAY, value) + + def deprecated_netcdf_member(self, name: str) -> Any: + """Return a member of the backing netCDF variable wrapper.""" + return getattr(self._variable, name) +``` + +`write_handle` is deliberately absent: it is a write concern and arrives in +Task 4. `NetCDFDatasetVariable` is therefore still abstract at the end of this +task — which is why every test above constructs it through `wrap`, and why +`TestStorageProperties.test_is_a_cf_dataset_variable` would fail if it were +not. Add the stub now, so the class is concrete: + +```python + def write_handle(self) -> Any: + """Return a picklable object supporting ``__setitem__``, for Dask stores.""" + raise NotImplementedError +``` + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/` +Expected: PASS — every test in the file, 0 failed, 0 skipped. + +- [ ] **Step 6: Check the module's docstrings satisfy numpydoc** + +Run: `pre-commit run numpydoc-validation --files lib/iris/fileformats/netcdf/_dataset.py` +Expected: Passed. Property setters need no docstring; everything else does, +one line, imperative, ending in a period. Fix anything it names. + +- [ ] **Step 7: Green check and commit** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/conftest.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py +pytest lib/iris/tests/unit/fileformats/netcdf/dataset/ +git add lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/ +git commit -m "$(cat <<'EOF' +Add NetCDFDatasetVariable, the read surface + +Implements the CFDatasetVariable interface over a thread-safe netCDF +variable wrapper, plus the netCDF-only members - the VLEN check, the +Xarray-bridge data array, and the backing wrapper itself - that the CF +layer used to reach for through getattr. + +Not wired to anything yet. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 3: `netcdf/_dataset.py` — the dataset, read surface + +**Files:** +- Modify: `lib/iris/fileformats/netcdf/_dataset.py` (append `NetCDFDataset`) +- Test: `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py` + +**Interfaces:** +- Consumes: `CFDataset` from Task 1; `NetCDFDatasetVariable` and + `_NetCDFAttributes` from Task 2; `sample_path` / `sample_location` / + `SAMPLE_DIMENSIONS` / `SAMPLE_GLOBALS` from Task 2's conftest. +- Produces: + - `NetCDFDataset(location, mode="r", *, netcdf_format=None, + warn_legacy_format=False)` — opens and owns a netCDF file. + - `NetCDFDataset.from_existing(dataset)` — borrows an already-open dataset + or emulator; `close()` on it is a no-op. Task 10 uses this from + `CFReader`; Task 12 uses it from `Saver`. + - `.dataset` (the backing `DatasetWrapper`) and `.write_lock`. Task 12 sets + `Saver.file_write_lock = self._dataset.write_lock`. + +- [ ] **Step 1: Write the failing tests** + +`lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.netcdf._dataset.NetCDFDataset`.""" + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDataset +from iris.fileformats.netcdf import _bytecoding_datasets, _dataset, _thread_safe_nc +from iris.warnings import IrisLoadWarning + +from .conftest import SAMPLE_DIMENSIONS, SAMPLE_GLOBALS + + +@pytest.fixture +def reader(sample_path): + with _dataset.NetCDFDataset(sample_path) as dataset: + yield dataset + + +class TestOpening: + def test_is_a_cf_dataset(self, reader): + assert isinstance(reader, CFDataset) + + def test_location_is_the_string_path(self, reader, sample_location): + assert reader.location == sample_location + + def test_default_mode_is_read(self, reader): + assert reader.mode == "r" + + def test_starts_open(self, reader): + assert reader.closed is False + + def test_close_is_idempotent(self, sample_path): + dataset = _dataset.NetCDFDataset(sample_path) + dataset.close() + assert dataset.closed is True + dataset.close() + assert dataset.closed is True + + def test_context_manager_closes(self, sample_path): + with _dataset.NetCDFDataset(sample_path) as dataset: + assert dataset.closed is False + assert dataset.closed is True + + def test_missing_file_raises(self, tmp_path): + with pytest.raises((FileNotFoundError, OSError)): + _dataset.NetCDFDataset(tmp_path / "nope.nc") + + def test_decoding_setting_chooses_the_wrapper(self, sample_path): + with _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ.context(True): + with _dataset.NetCDFDataset(sample_path) as dataset: + assert isinstance(dataset.dataset, _bytecoding_datasets.EncodedDataset) + with _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ.context(False): + with _dataset.NetCDFDataset(sample_path) as dataset: + assert not isinstance( + dataset.dataset, _bytecoding_datasets.EncodedDataset + ) + assert isinstance(dataset.dataset, _thread_safe_nc.DatasetWrapper) + + +class TestContents: + def test_variables(self, reader, sample_location): + assert sorted(reader.variables) == ["air_temperature", "height", "label"] + air = reader.variables["air_temperature"] + assert isinstance(air, _dataset.NetCDFDatasetVariable) + assert air.name == "air_temperature" + assert air.location == sample_location + + def test_variables_are_built_once(self, reader): + assert ( + reader.variables["air_temperature"] + is reader.variables["air_temperature"] + ) + + def test_variables_share_one_write_lock(self, reader): + # _dask_locks.get_worker_lock() returns a FRESH threading.Lock under + # the threaded scheduler, so a per-variable call would hand out locks + # that exclude nothing. + air = reader.variables["air_temperature"] + height = reader.variables["height"] + assert air._write_lock is height._write_lock + assert air._write_lock is reader.write_lock + + def test_dimensions_are_names_and_lengths(self, reader): + assert dict(reader.dimensions) == SAMPLE_DIMENSIONS + + def test_unlimited_dimension_reports_its_current_length(self, reader): + # Nothing was written along "record", so it is zero-length - which is + # what netCDF4 reports, and what saver.py's membership tests need. + assert reader.dimensions["record"] == 0 + assert "record" in reader.dimensions + + def test_global_attributes(self, reader): + assert dict(reader.attributes) == SAMPLE_GLOBALS + + def test_global_attributes_are_built_once(self, reader): + assert reader.attributes is reader.attributes + + +class TestBorrowing: + def test_from_existing_wraps_an_open_dataset(self, sample_path, sample_location): + raw = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + assert dataset.dataset is raw + assert dataset.location == sample_location + assert sorted(dataset.variables) == [ + "air_temperature", + "height", + "label", + ] + finally: + raw.close() + + def test_from_existing_does_not_close_what_it_borrowed(self, sample_path): + raw = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + dataset.close() + assert dataset.closed is True + # Still usable: the borrower released nothing. + assert raw.isopen() + finally: + raw.close() + + def test_from_existing_wraps_a_bare_netcdf4_dataset(self, sample_path): + # What the Xarray bridge hands iris.save / CFReader: an object with + # the netCDF4 API but no thread-safe wrapper around it. + import netCDF4 + + raw = netCDF4.Dataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + assert isinstance( + dataset.dataset, _bytecoding_datasets.EncodedDataset + ) + assert dataset.dataset._contained_instance is raw + finally: + raw.close() + + +class TestAutoChartostring: + """Iris decodes byte data itself, so netCDF4 must not do it first. + + CFReader turned this off on every dataset it opened (_reader.py:182). + The dataset now does it, so no caller has to remember. + """ + + def test_turned_off_on_an_opened_dataset(self, sample_path, mocker): + spy = mocker.spy(_thread_safe_nc.DatasetWrapper, "set_auto_chartostring") + with _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ.context(False): + with _dataset.NetCDFDataset(sample_path): + pass + spy.assert_called_once_with(mocker.ANY, False) + + def test_turned_off_on_a_borrowed_dataset(self, sample_path, mocker): + raw = _thread_safe_nc.DatasetWrapper(sample_path, mode="r") + spy = mocker.spy(raw, "set_auto_chartostring") + try: + _dataset.NetCDFDataset.from_existing(raw) + spy.assert_called_once_with(False) + finally: + raw.close() + + def test_an_encoded_dataset_blocks_it_rather_than_forwarding(self, sample_path): + # EncodedDataset does its own decoding, so the call is inert there - + # which is why making it unconditionally is safe. + with _dataset.NetCDFDataset(sample_path) as dataset: + assert isinstance(dataset.dataset, _bytecoding_datasets.EncodedDataset) + with pytest.raises(TypeError, match="not supported"): + dataset.dataset.set_auto_chartostring(True) + + +class TestLegacyFormatWarning: + def test_silent_by_default(self, tmp_path, sample_path): + legacy = _write_netcdf3(tmp_path) + with warnings.catch_warnings(): + warnings.simplefilter("error") + with _dataset.NetCDFDataset(legacy): + pass + + def test_warns_when_asked(self, tmp_path): + legacy = _write_netcdf3(tmp_path) + with pytest.warns(IrisLoadWarning, match="nccopy"): + with _dataset.NetCDFDataset(legacy, warn_legacy_format=True): + pass + + def test_does_not_warn_for_netcdf4(self, sample_path): + with warnings.catch_warnings(): + warnings.simplefilter("error") + with _dataset.NetCDFDataset(sample_path, warn_legacy_format=True): + pass + + +def _write_netcdf3(tmp_path): + path = tmp_path / "legacy.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w", format="NETCDF3_CLASSIC") + dataset.createDimension("time", 2) + variable = dataset.createVariable("time", "f4", ("time",)) + variable[:] = np.arange(2, dtype="f4") + dataset.close() + return path +``` + +Add `import warnings` at the top of the file, beside `numpy`. + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py` +Expected: collection error — +`AttributeError: module 'iris.fileformats.netcdf._dataset' has no attribute 'NetCDFDataset'` + +- [ ] **Step 3: Write the dataset's read surface** + +Append to `lib/iris/fileformats/netcdf/_dataset.py`: + +```python +#: The formats that make CF loading slow, and that the user can convert away +#: from with "nccopy". +_LEGACY_FORMATS = ("NETCDF3_CLASSIC", "NETCDF3_64BIT") + + +class NetCDFDataset(CFDataset): + """A netCDF file, presented through the CF interface.""" + + def __init__( + self, + location, + mode: str = "r", + *, + netcdf_format=None, + warn_legacy_format: bool = False, + ): + """Open a netCDF file. + + Parameters + ---------- + location : str or :class:`pathlib.Path` + The file's path or URL. + mode : str, default="r" + The netCDF4 open mode. + netcdf_format : str, optional + The netCDF format to create, when writing. + warn_legacy_format : bool, default=False + Whether to warn that a netCDF3 file would load faster if + converted. Only loading asks for this; the loader is where the + user can act on it. + + """ + # Set first, so that __del__ and close() are safe if opening fails. + self._dataset = None + self._owned = True + self._closed = False + self._variables: dict[str, NetCDFDatasetVariable] | None = None + self._attributes: _NetCDFAttributes | None = None + self._write_lock = None + + self._location = str(location) + self._mode = mode + + if mode == "r" and not _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ: + # The user has turned string decoding off for reads. Writing has + # no such switch: the saver always encodes. + dataset_class = _thread_safe_nc.DatasetWrapper + else: + dataset_class = _bytecoding_datasets.EncodedDataset + + # netCDF4.Dataset validates "format" against a fixed list and rejects + # None, so omit it entirely rather than passing the default through. + extra = {} if netcdf_format is None else {"format": netcdf_format} + self._dataset = dataset_class(location, mode=mode, **extra) + + if warn_legacy_format and self._dataset.file_format in _LEGACY_FORMATS: + warnings.warn( + "Optimise CF-netCDF loading by converting data from NetCDF3 " + 'to NetCDF4 file format using the "nccopy" command.', + category=iris.warnings.IrisLoadWarning, + ) + + # Turn off *any* automatic decoding by netCDF4 itself. Iris decodes + # byte data on its own terms, in _bytecoding_datasets. Inert on an + # EncodedDataset, which blocks the call; real on a DatasetWrapper. + self._dataset.set_auto_chartostring(False) + + @classmethod + def from_existing(cls, dataset) -> "NetCDFDataset": + """Wrap an already-open netCDF dataset, without taking ownership of it. + + ``dataset`` may be a thread-safe wrapper, a bare + :class:`netCDF4.Dataset`, or any object emulating one - the Xarray + bridge passes the last of these. :meth:`close` will not release it, + because whoever opened it is still responsible for it. + + """ + instance = cls.__new__(cls) + instance._owned = False + instance._closed = False + instance._variables = None + instance._attributes = None + instance._write_lock = None + instance._mode = "r+" + + if not hasattr(dataset, "THREAD_SAFE_FLAG"): + # The wrappers forbid re-wrapping, so only wrap what is not one. + dataset = _bytecoding_datasets.EncodedDataset.from_existing(dataset) + instance._dataset = dataset + instance._dataset.set_auto_chartostring(False) + instance._location = str(dataset.filepath()) + return instance + + @property + def location(self) -> str: + """The file's path or URL.""" + return self._location + + @property + def mode(self) -> str: + """The mode the file was opened in.""" + return self._mode + + @property + def closed(self) -> bool: + """Whether :meth:`close` has already released the file.""" + return self._closed + + @property + def variables(self) -> Mapping: + """The file's variables, by name.""" + if self._variables is None: + self._variables = { + name: NetCDFDatasetVariable( + variable, self._location, write_lock=self.write_lock + ) + for name, variable in self._dataset.variables.items() + } + return self._variables + + @property + def dimensions(self) -> Mapping: + """The file's dimension lengths, by name. + + An unlimited dimension reports the number of records written so far, + which is what ``len()`` of a netCDF4 dimension gives and what every + caller in Iris - all of them membership tests - needs. + + """ + return { + name: len(dimension) + for name, dimension in self._dataset.dimensions.items() + } + + @property + def attributes(self) -> MutableMapping: + """The file's global attributes.""" + if self._attributes is None: + self._attributes = _NetCDFAttributes(self._dataset) + return self._attributes + + def sync(self) -> None: + """Flush buffered writes to the file.""" + self._dataset.sync() + + def finalise(self) -> None: + """Do nothing: a netCDF file needs no completion step. + + Kept so that callers can be written once. Zarr's consolidated metadata + is what this exists for - see finding F6. + + """ + + def close(self) -> None: + """Close the file, if this object opened it.""" + if self._owned and self._dataset is not None and not self._closed: + self._dataset.close() + self._closed = True + + def __repr__(self) -> str: + """Return a string representation.""" + return f"{self.__class__.__name__}({self.location!r}, mode={self.mode!r})" + + # netCDF-only members below. + + @property + def dataset(self): + """The backing thread-safe netCDF dataset wrapper.""" + return self._dataset + + @property + def write_lock(self): + """The lock every worker writing to this file must hold. + + One per dataset, because under the threaded scheduler + :func:`iris.fileformats.netcdf._dask_locks.get_worker_lock` returns a + new :class:`threading.Lock` on each call, and two such locks exclude + nothing. + + """ + if self._write_lock is None: + self._write_lock = _dask_locks.get_worker_lock(self._location) + return self._write_lock +``` + +Extend the module's imports to match: + +```python +from collections.abc import Iterator, Mapping, MutableMapping +import warnings + +import iris.warnings + +from . import _bytecoding_datasets, _dask_locks, _thread_safe_nc +``` + +`create_dimension` and `create_variable` are still abstract at this point, so +the class cannot be instantiated. Add them as stubs now — Task 4 fills them +in, driven by its own tests: + +```python + def create_dimension(self, name: str, size: int | None) -> None: + """Declare a dimension of the given length.""" + raise NotImplementedError + + def create_variable( + self, name: str, dtype, dimensions=(), *, fill_value=None, **encoding + ) -> NetCDFDatasetVariable: + """Create and return a new variable.""" + raise NotImplementedError +``` + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/` +Expected: PASS — every test in both files, 0 failed, 0 skipped. + +- [ ] **Step 5: Confirm the interface is still one-way** + +```bash +python - <<'PY' +src = open('lib/iris/fileformats/cf/dataset.py').read() +assert 'netcdf' not in src and 'netCDF4' not in src +import iris.fileformats.netcdf._dataset as m +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable +assert issubclass(m.NetCDFDataset, CFDataset) +assert issubclass(m.NetCDFDatasetVariable, CFDatasetVariable) +print('ok') +PY +``` +Expected: `ok`. + +- [ ] **Step 6: Green check and commit** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py +pytest lib/iris/tests/unit/fileformats/netcdf/dataset/ +git add lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py +git commit -m "$(cat <<'EOF' +Add NetCDFDataset, the read surface + +Opens a netCDF file, or borrows an already-open one, and presents its +variables, dimensions and global attributes through the CF interface. + +The NetCDF3 "use nccopy" warning moves here from CFReader, behind an +explicit warn_legacy_format flag: the file format is netCDF vocabulary, +and CFReader is no longer allowed to know it. + +Not wired to anything yet. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 4: `netcdf/_dataset.py` — the write surface + +**Files:** +- Modify: `lib/iris/fileformats/netcdf/_dataset.py` +- Modify: `lib/iris/fileformats/netcdf/saver.py:271-296` (`_bytes_if_ascii` moves out) +- Test: `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py` + +**Interfaces:** +- Consumes: `NetCDFDataset` and `NetCDFDatasetVariable` from Tasks 2 and 3. +- Produces: + - `NetCDFDataset.create_dimension(name, size)` — `size=None` is unlimited. + - `NetCDFDataset.create_variable(name, dtype, dimensions=(), *, + fill_value=None, **encoding)` returning a `NetCDFDatasetVariable`. + `dimensions` defaults to `()` so that `saver.py:2080`'s dimensionless + grid-mapping variable needs no argument (finding F4). `**encoding` is + passed straight to `netCDF4.createVariable`: `zlib`, `complevel`, + `shuffle`, `chunksizes`, `compression`, and the rest. + - `NetCDFDatasetVariable.write_handle()` returning an + `EncodedNetCDFWriteProxy` (or a `NetCDFWriteProxy` for an unencoded + variable). Task 12 replaces `saver.py:2641`'s direct construction with it. + - `_dataset._bytes_if_ascii(value)`, moved here from `saver.py:271`. + `saver.py` imports it back for its own `_setncattr`, which Task 12 deletes. + +- [ ] **Step 1: Write the failing tests for dimensions, variables and writes** + +`lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for the write surface of the netCDF CFDataset implementation.""" + +import numpy as np +import pytest + +from iris.fileformats.netcdf import _bytecoding_datasets, _dataset, _thread_safe_nc + + +@pytest.fixture +def path(tmp_path): + return tmp_path / "written.nc" + + +@pytest.fixture +def writer(path): + with _dataset.NetCDFDataset(path, mode="w", netcdf_format="NETCDF4") as dataset: + yield dataset + + +@pytest.fixture +def grid(writer): + """A 2x3 grid, ready for variables to be created against.""" + writer.create_dimension("y", 3) + writer.create_dimension("x", 2) + return writer + + +class TestCreateDimension: + def test_appears_with_its_length(self, writer): + writer.create_dimension("x", 5) + assert writer.dimensions["x"] == 5 + + def test_none_means_unlimited(self, writer): + # saver.py:829 passes None for a dimension the user asked to make + # unlimited. It reads back as zero-length until records are written. + writer.create_dimension("t", None) + assert writer.dimensions["t"] == 0 + assert writer.dataset.dimensions["t"].isunlimited() + + +class TestCreateVariable: + def test_returns_a_dataset_variable(self, grid): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert isinstance(variable, _dataset.NetCDFDatasetVariable) + assert variable.name == "air" + assert variable.dimensions == ("y", "x") + assert variable.shape == (3, 2) + assert variable.dtype == np.dtype("f4") + + def test_appears_in_the_dataset(self, grid): + created = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert grid.variables["air"].name == created.name + + def test_dimensions_default_to_scalar(self, writer): + # saver.py:2080 creates a grid-mapping variable with no dimensions at + # all, and passes only a name and a dtype - see finding F4. + variable = writer.create_variable("grid", np.int32) + assert variable.dimensions == () + assert variable.shape == () + + def test_dimensions_may_be_a_list(self, grid): + # saver.py:1938 and :1983 pass a list, not a tuple. + variable = grid.create_variable("air", np.dtype("f4"), ["y", "x"]) + assert variable.dimensions == ("y", "x") + + def test_fill_value_becomes_an_attribute(self, grid): + variable = grid.create_variable( + "air", np.dtype("f4"), ("y", "x"), fill_value=-1.0 + ) + assert variable.fill_value == -1.0 + assert variable.attributes["_FillValue"] == -1.0 + + def test_encoding_is_passed_through(self, grid): + variable = grid.create_variable( + "air", np.dtype("f4"), ("y", "x"), zlib=True, chunksizes=(1, 2) + ) + assert variable.chunking == (1, 2) + + def test_shares_the_datasets_write_lock(self, grid): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert variable._write_lock is grid.write_lock + + +class TestWritingData: + def test_setitem_round_trip(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + payload = np.arange(6, dtype="f4").reshape(3, 2) + variable[:] = payload + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + np.testing.assert_array_equal(reader.variables["air"][:], payload) + + def test_write_handle_writes_after_the_dataset_is_closed(self, grid, path): + # This is what a Dask worker gets as a da.store target, long after the + # Saver's own handle on the file has gone. + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + handle = variable.write_handle() + grid.close() + + payload = np.arange(6, dtype="f4").reshape(3, 2) + handle[:] = payload + + with _dataset.NetCDFDataset(path) as reader: + np.testing.assert_array_equal(reader.variables["air"][:], payload) + + def test_write_handle_matches_the_variable_encoding(self, grid): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + # The dataset was opened for writing, so its variables are encoded; + # an unencoded proxy here would silently skip string encoding. + assert isinstance( + variable.write_handle(), _bytecoding_datasets.EncodedNetCDFWriteProxy + ) + + def test_write_handle_of_an_unencoded_variable(self, path): + raw = _thread_safe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + try: + raw.createDimension("x", 2) + raw.createVariable("air", "f4", ("x",)) + variable = _dataset.NetCDFDatasetVariable( + raw.variables["air"], str(path), write_lock=None + ) + assert isinstance( + variable.write_handle(), _thread_safe_nc.NetCDFWriteProxy + ) + finally: + raw.close() + + +class TestSyncAndFinalise: + def test_sync_flushes(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable[:] = np.zeros((3, 2), dtype="f4") + grid.sync() + # Readable while the writer is still open, because sync() flushed. + with _dataset.NetCDFDataset(path) as reader: + assert reader.variables["air"].shape == (3, 2) + + def test_finalise_is_a_no_op(self, grid): + assert grid.finalise() is None + assert grid.closed is False +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py` +Expected: FAIL — `NotImplementedError` from the `create_dimension`, +`create_variable` and `write_handle` stubs left by Tasks 2 and 3. + +- [ ] **Step 3: Implement the write surface** + +In `lib/iris/fileformats/netcdf/_dataset.py`, replace +`NetCDFDatasetVariable.write_handle`'s stub: + +```python + def write_handle(self) -> Any: + """Return a picklable object supporting ``__setitem__``, for Dask stores. + + It carries the file path and variable name rather than the open file, + reopening on each write, so that a worker can use it after the saver + that created it has closed its own handle. + + """ + if isinstance(self._variable, _bytecoding_datasets.EncodedVariable): + proxy_class = _bytecoding_datasets.EncodedNetCDFWriteProxy + else: + # Only reachable for a dataset opened without string encoding; + # the saver always encodes. + proxy_class = _thread_safe_nc.NetCDFWriteProxy + return proxy_class(self._location, self._variable, self._write_lock) +``` + +and replace `NetCDFDataset`'s two stubs: + +```python + def create_dimension(self, name: str, size: int | None) -> None: + """Declare a dimension of the given length, or unlimited for ``None``.""" + self._dataset.createDimension(name, size) + + def create_variable( + self, name: str, dtype, dimensions=(), *, fill_value=None, **encoding + ) -> NetCDFDatasetVariable: + """Create and return a new variable. + + ``**encoding`` reaches :meth:`netCDF4.Dataset.createVariable` + unaltered: ``compression``, ``zlib``, ``complevel``, ``shuffle``, + ``chunksizes``, ``least_significant_digit`` and the rest. + + """ + variable = self._dataset.createVariable( + name, dtype, tuple(dimensions), fill_value=fill_value, **encoding + ) + wrapped = NetCDFDatasetVariable( + variable, self._location, write_lock=self.write_lock + ) + if self._variables is not None: + # Keep an already-materialised mapping in step, rather than + # leaving it stale for the rest of the dataset's life. + self._variables[name] = wrapped + return wrapped +``` + +netCDF4 rejects `fill_value=None` for some dtypes only when combined with +other options; it is its own documented default, so passing it through +unconditionally is correct and keeps one code path. + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/` +Expected: PASS — every test in all three files, 0 failed, 0 skipped. + +- [ ] **Step 5: Commit the write surface** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py +pytest lib/iris/tests/unit/fileformats/netcdf/dataset/ +git add lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py +git commit -m "$(cat <<'EOF' +Add the netCDF CFDataset write surface + +create_dimension, create_variable and write_handle(). write_handle is +the seam that lets a Dask worker write into a variable after the saver +has closed the file, and is where the Zarr implementation will return a +zarr.Array instead of a reopen-on-write proxy. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 6: Write the failing test for attribute write-through** + +Append to `test_NetCDFDataset__write.py`: + +```python +class TestAttributeWrites: + def test_variable_attribute_write_through(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["units"] = "K" + assert variable.attributes["units"] == "K" + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.variables["air"].attributes["units"] == "K" + + def test_global_attribute_write_through(self, writer, path): + writer.attributes["Conventions"] = "CF-1.7" + assert writer.attributes["Conventions"] == "CF-1.7" + writer.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.attributes["Conventions"] == "CF-1.7" + + def test_attribute_deletion(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["units"] = "K" + del variable.attributes["units"] + assert "units" not in variable.attributes + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + assert "units" not in reader.variables["air"].attributes + + def test_ascii_value_is_written_as_bytes(self, grid): + # _bytes_if_ascii, moved here from saver.py. netCDF4 writes a bytes + # value as NC_CHAR; the coercion is what makes every string attribute + # Iris writes take that type, whatever the file format. + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["units"] = "K" + assert variable.variable.getncattr("units") == "K" + assert _dataset._bytes_if_ascii("K") == b"K" + + def test_non_string_values_pass_through(self, grid): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["valid_min"] = np.float32(-3.5) + assert variable.attributes["valid_min"] == np.float32(-3.5) + assert _dataset._bytes_if_ascii(3) == 3 +``` + +- [ ] **Step 7: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py -k Attribute` +Expected: FAIL — +`AttributeError: module 'iris.fileformats.netcdf._dataset' has no attribute '_bytes_if_ascii'` + +- [ ] **Step 8: Move `_bytes_if_ascii` and apply it on write** + +Add to `lib/iris/fileformats/netcdf/_dataset.py`, above `_NetCDFAttributes`: + +```python +def _bytes_if_ascii(value): + """Return an ASCII string as bytes, and anything else unchanged. + + netCDF4 stores a bytes attribute as NC_CHAR. Coercing on the way in is + what keeps Iris's string attributes to that type across file formats, + rather than leaving it to netCDF4's own str handling. + + """ + if isinstance(value, str): + try: + return value.encode(encoding="ascii") + except (AttributeError, UnicodeEncodeError): + pass + return value +``` + +and apply it in `_NetCDFAttributes.__setitem__`: + +```python + def __setitem__(self, key: str, value: Any) -> None: + """Set ``key``'s value, here and in the netCDF object.""" + # Coerce on the way out only. Caching the coerced value would make a + # just-written attribute read back as bytes, where the same attribute + # read from a file reads back as str. See finding F12. + self._target.setncattr(key, _bytes_if_ascii(value)) + self._values[key] = value +``` + +The comment is load-bearing, and the fact it rests on is worth seeing for +yourself before you trust it — netCDF4 writes the `bytes` as `NC_CHAR` and +hands back a `str` **[verified]**: + +```python +>>> ds.setncattr("nm", b"hello"); ds.getncattr("nm") +'hello' +``` + +So a cache holding the *coerced* value disagrees with the file it mirrors: +`attributes["units"]` would be `b"K"` on the dataset that just wrote it and +`"K"` on one that reads it back. Two saver comparisons need the `str` — +`cf_var.attributes["formula_terms"] != formula_terms` (`saver.py:1184`) and +`" ".join(coords)` (`:1213`) — and both fail silently, as a spurious +inequality and a `TypeError` respectively, against `bytes`. + +Step 6 already wrote the test: `test_variable_attribute_write_through`'s +`assert variable.attributes["units"] == "K"` is on the *writing* dataset, +before the file is reopened. Write `value = _bytes_if_ascii(value)` on its +own line above the two, as the obvious reading of Step 6 suggests, and that +assertion is what goes red. + +Then, in `lib/iris/fileformats/netcdf/saver.py`, delete the `_bytes_if_ascii` +definition at line 271 and import it instead, leaving `_setncattr` otherwise +untouched: + +```python +from ._dataset import _bytes_if_ascii +``` + +Task 12 removes `_setncattr` and this import together. Check nothing else +imported it from `saver`: + +```bash +grep -rn "_bytes_if_ascii" lib/ benchmarks/ docs/ +``` + +- [ ] **Step 9: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/` +Expected: PASS — 0 failed, 1 skipped. That one skip is +`attribute_handlers/test_GribParamHandler.py` wanting `iris_grib`, and it is +0 if you have `iris_grib` installed (§6.2). Any *other* skip is a test that +stopped being collected; treat it as a failure. + +- [ ] **Step 10: Review Focus 5 — a non-ASCII attribute value** + +`_bytes_if_ascii` is a `try`/`except` whose failure path is the interesting +one: a value that will not encode as ASCII is left as `str`. Every degree +sign, every accented place name in a `long_name` takes that path, and the +relocated coercion has to keep taking it. Append: + +```python +class TestNonAsciiAttributes: + """Review Focus 5. The coercion's except branch, which is the common one + for real-world metadata: degree signs, accented names, superscripts.""" + + @pytest.mark.parametrize( + "value", + ["degC \N{DEGREE SIGN}", "na\N{LATIN SMALL LETTER I WITH DIAERESIS}ve", + "m s\N{SUPERSCRIPT MINUS}\N{SUPERSCRIPT ONE}"], + ) + def test_round_trips_unchanged(self, grid, path, value): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["long_name"] = value + assert variable.attributes["long_name"] == value + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.variables["air"].attributes["long_name"] == value + + def test_is_not_coerced_to_bytes(self): + value = "degC \N{DEGREE SIGN}" + assert _dataset._bytes_if_ascii(value) is value + + def test_global_non_ascii_round_trips(self, writer, path): + value = "Produced at M\N{LATIN SMALL LETTER E WITH ACUTE}t\N{LATIN SMALL LETTER E WITH ACUTE}o" + writer.attributes["institution"] = value + writer.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.attributes["institution"] == value +``` + +- [ ] **Step 11: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/` +Expected: PASS. These should pass on the implementation as written; if any +fails, the coercion is over-reaching and the `except` clause is what to look +at, not the test. + +Then the saver's own suite, which is the largest consumer of `_bytes_if_ascii`: + +Run: `pytest -n auto lib/iris/tests/unit/fileformats/netcdf/ lib/iris/tests/integration/` +Expected: 0 failed, and the same skip count as before this task. + +- [ ] **Step 12: Green check and commit** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/fileformats/netcdf/saver.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py +git add lib/iris/fileformats/netcdf/_dataset.py lib/iris/fileformats/netcdf/saver.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py +git commit -m "$(cat <<'EOF' +Write netCDF attributes through the CFDataset interface + +_NetCDFAttributes.__setitem__ applies the ASCII coercion that saver.py's +_setncattr applied, so the coercion stays on the netCDF path - as the +spec requires - instead of following CF attributes into generic code. + +_bytes_if_ascii moves from saver.py to _dataset.py; saver.py imports it +back until its own _setncattr goes. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 5: the shared `CFDataset` contract test + +Spec section 6: *"The `CFDataset` contract gets one shared test body run +against both implementations."* This is that body. PR 4 adds +`TestZarrDatasetContract(CFDatasetContract)` and inherits every test below +without writing one of them again. + +**Files:** +- Create: `lib/iris/tests/unit/fileformats/cf/dataset/contract.py` +- Test: `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py` + +**Interfaces:** +- Consumes: `CFDataset`, `CFDatasetVariable` (Task 1); `NetCDFDataset` + (Tasks 3 and 4). +- Produces: + - `CFDatasetContract` — a test body with **no** `Test` prefix, so pytest + does not collect it where it is defined. A subclass must supply three + fixtures: `readable`, `writable`, and `reopen` (a zero-argument callable + returning a fresh read-mode dataset over what `writable` wrote). + - `populate(dataset)` — fills an empty writable dataset with the canonical + contents, using only the `CFDataset` API. + - the constants `CONTRACT_DIMENSIONS`, `CONTRACT_DATA`, + `CONTRACT_VARIABLE_ATTRS`, `CONTRACT_GLOBALS`. + +- [ ] **Step 1: Write the contract body** + +`lib/iris/tests/unit/fileformats/cf/dataset/contract.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""One test body, run against every :class:`CFDataset` implementation. + +Subclass :class:`CFDatasetContract` in a module named for the implementation, +supply the three fixtures it declares, and every test here runs against it. +The class deliberately has no ``Test`` prefix, so pytest does not collect it +where it is written, only where it is subclassed. + +Spec section 6: the interface is only worth having if both sides of it agree, +and agreement is cheapest to check by asking the same questions twice. +""" + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable + +#: The canonical dataset every implementation is checked against. +CONTRACT_DIMENSIONS = {"y": 3, "x": 2} +CONTRACT_DATA = np.arange(6, dtype="f8").reshape(3, 2) +CONTRACT_VARIABLE_ATTRS = {"units": "K", "long_name": "surface temperature"} +CONTRACT_GLOBALS = {"Conventions": "CF-1.7", "title": "contract"} + + +def populate(dataset): + """Fill an empty writable dataset with the canonical contents. + + Uses only the CFDataset API, so this doubles as the write half of the + contract: an implementation that cannot build this cannot be used to save. + """ + for name, size in CONTRACT_DIMENSIONS.items(): + dataset.create_dimension(name, size) + for name, value in CONTRACT_GLOBALS.items(): + dataset.attributes[name] = value + + variable = dataset.create_variable( + "air_temperature", np.dtype("f8"), tuple(CONTRACT_DIMENSIONS), fill_value=-1.0 + ) + for name, value in CONTRACT_VARIABLE_ATTRS.items(): + variable.attributes[name] = value + variable[:] = CONTRACT_DATA + + # A dimensionless variable: what a grid mapping is, and the one shape a + # CFDatasetVariable must handle without a leading dimension. + scalar = dataset.create_variable("grid", np.dtype("i4")) + scalar.attributes["grid_mapping_name"] = "latitude_longitude" + scalar[()] = 0 + return dataset + + +class CFDatasetContract: + """What every CFDataset implementation must do, whatever it stores into.""" + + @pytest.fixture + def readable(self): + """Return an open read-mode dataset holding what populate() writes.""" + raise NotImplementedError("supply a 'readable' fixture") + + @pytest.fixture + def writable(self): + """Return an open, empty, write-mode dataset.""" + raise NotImplementedError("supply a 'writable' fixture") + + @pytest.fixture + def reopen(self): + """Return a callable giving a fresh read-mode dataset over 'writable'.""" + raise NotImplementedError("supply a 'reopen' fixture") + + # -- The dataset itself ------------------------------------------------ + + def test_is_a_cf_dataset(self, readable): + assert isinstance(readable, CFDataset) + + def test_location_is_a_non_empty_string(self, readable): + assert isinstance(readable.location, str) + assert readable.location + + def test_mode_is_a_read_mode(self, readable): + assert readable.mode in ("r", "r+", "a") + + def test_starts_open(self, readable): + assert readable.closed is False + + def test_close_then_closed(self, writable): + writable.close() + assert writable.closed is True + + def test_close_is_idempotent(self, writable): + writable.close() + writable.close() + assert writable.closed is True + + def test_context_manager_returns_self_and_closes(self, writable): + with writable as entered: + assert entered is writable + assert writable.closed is True + + def test_dimensions(self, readable): + assert dict(readable.dimensions) == CONTRACT_DIMENSIONS + + def test_global_attributes(self, readable): + assert dict(readable.attributes) == CONTRACT_GLOBALS + + def test_variable_names(self, readable): + assert sorted(readable.variables) == ["air_temperature", "grid"] + + # -- A variable -------------------------------------------------------- + + @pytest.fixture + def variable(self, readable): + return readable.variables["air_temperature"] + + @pytest.fixture + def scalar(self, readable): + return readable.variables["grid"] + + def test_variable_is_a_cf_dataset_variable(self, variable): + assert isinstance(variable, CFDatasetVariable) + + def test_variable_name(self, variable): + assert variable.name == "air_temperature" + + def test_variable_location_matches_its_dataset(self, variable, readable): + assert variable.location == readable.location + + def test_variable_dimensions(self, variable): + assert variable.dimensions == tuple(CONTRACT_DIMENSIONS) + + def test_variable_shape(self, variable): + assert variable.shape == tuple(CONTRACT_DIMENSIONS.values()) + + def test_variable_dtype(self, variable): + assert variable.dtype == np.dtype("f8") + + def test_variable_size(self, variable): + assert variable.size == CONTRACT_DATA.size + + def test_variable_ndim(self, variable): + assert variable.ndim == CONTRACT_DATA.ndim + + def test_variable_len(self, variable): + assert len(variable) == CONTRACT_DATA.shape[0] + + def test_variable_data(self, variable): + np.testing.assert_array_equal(variable[:], CONTRACT_DATA) + + def test_variable_data_indexed(self, variable): + np.testing.assert_array_equal(variable[1], CONTRACT_DATA[1]) + + def test_variable_attributes(self, variable): + for name, value in CONTRACT_VARIABLE_ATTRS.items(): + assert variable.attributes[name] == value + + def test_variable_attributes_omit_unset_names(self, variable): + assert "nonesuch" not in variable.attributes + with pytest.raises(KeyError): + variable.attributes["nonesuch"] + + def test_variable_fill_value(self, variable): + assert variable.fill_value == -1.0 + + def test_variable_chunking_is_none_or_a_shape(self, variable): + chunking = variable.chunking + assert chunking is None or ( + isinstance(chunking, tuple) + and len(chunking) == len(variable.shape) + and all(isinstance(size, int) for size in chunking) + ) + + def test_scalar_variable(self, scalar): + assert scalar.dimensions == () + assert scalar.shape == () + with pytest.raises(TypeError, match="unsized"): + len(scalar) + + # -- Writing ----------------------------------------------------------- + + def test_populate_then_read_back(self, writable, reopen): + populate(writable) + writable.close() + with reopen() as reader: + np.testing.assert_array_equal( + reader.variables["air_temperature"][:], CONTRACT_DATA + ) + assert dict(reader.dimensions) == CONTRACT_DIMENSIONS + assert dict(reader.attributes) == CONTRACT_GLOBALS + + def test_created_variable_is_visible_on_the_dataset(self, writable): + writable.create_dimension("x", 2) + created = writable.create_variable("thing", np.dtype("f4"), ("x",)) + assert writable.variables["thing"].name == created.name + + def test_create_variable_without_dimensions(self, writable): + # Finding F4: saver.py:2080 supplies only a name and a dtype. + variable = writable.create_variable("grid", np.dtype("i4")) + assert variable.dimensions == () + + def test_setitem_then_getitem(self, writable): + writable.create_dimension("x", 3) + variable = writable.create_variable("thing", np.dtype("f4"), ("x",)) + variable[:] = [1.0, 2.0, 3.0] + np.testing.assert_array_equal(variable[:], [1.0, 2.0, 3.0]) + + def test_attribute_write_through(self, writable, reopen): + populate(writable) + writable.variables["air_temperature"].attributes["comment"] = "added" + writable.close() + with reopen() as reader: + assert reader.variables["air_temperature"].attributes["comment"] == "added" + + def test_write_handle_works_after_close(self, writable, reopen): + populate(writable) + handle = writable.variables["air_temperature"].write_handle() + writable.close() + + payload = CONTRACT_DATA * 10 + handle[:] = payload + with reopen() as reader: + np.testing.assert_array_equal( + reader.variables["air_temperature"][:], payload + ) + + def test_sync_and_finalise_are_callable(self, writable): + populate(writable) + assert writable.sync() is None + assert writable.finalise() is None + assert writable.closed is False +``` + +- [ ] **Step 2: Write the netCDF subclass** + +`lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Run the shared CFDataset contract against the netCDF implementation.""" + +import pytest + +from iris.fileformats.netcdf import _dataset +from iris.tests.unit.fileformats.cf.dataset.contract import ( + CFDatasetContract, + populate, +) + + +class TestNetCDFDatasetContract(CFDatasetContract): + """The netCDF side of the contract. PR 4 adds the Zarr side.""" + + @pytest.fixture + def readable(self, tmp_path): + path = tmp_path / "contract.nc" + with _dataset.NetCDFDataset( + path, mode="w", netcdf_format="NETCDF4" + ) as dataset: + populate(dataset) + with _dataset.NetCDFDataset(path) as dataset: + yield dataset + + @pytest.fixture + def writable(self, tmp_path): + dataset = _dataset.NetCDFDataset( + tmp_path / "written.nc", mode="w", netcdf_format="NETCDF4" + ) + yield dataset + dataset.close() + + @pytest.fixture + def reopen(self, tmp_path): + def _reopen(): + return _dataset.NetCDFDataset(tmp_path / "written.nc") + + return _reopen +``` + +The `writable` fixture closes unconditionally on teardown, which is why +`CFDataset.close()` has to be idempotent — several contract tests close it +themselves first. + +- [ ] **Step 3: Run the contract to verify it fails** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py` +Expected: collection error — +`ModuleNotFoundError: No module named 'iris.tests.unit.fileformats.cf.dataset.contract'` + +- [ ] **Step 4: Run the contract to verify it passes** + +Once both files exist: + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py -v` +Expected: PASS — every test, 0 failed, 0 skipped. + +Anything that fails here is a real disagreement between Tasks 1–4 and the +interface as written; fix the implementation, not the contract. The contract +is what PR 4 inherits, so a test loosened here to fit netCDF stops being a +check on Zarr. + +- [ ] **Step 5: Confirm the contract body is not collected where it lives** + +```bash +pytest lib/iris/tests/unit/fileformats/cf/dataset/ --collect-only -q | grep -c contract +``` +Expected: `0`. `contract.py` does not match `python_files`, and +`CFDatasetContract` does not match `python_classes`; both must hold, because +the fixtures raise `NotImplementedError` and would report as errors. + +- [ ] **Step 6: Green check and commit** + +```bash +pre-commit run --files lib/iris/tests/unit/fileformats/cf/dataset/contract.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py +pytest lib/iris/tests/unit/fileformats/cf/ lib/iris/tests/unit/fileformats/netcdf/dataset/ +git add lib/iris/tests/unit/fileformats/cf/dataset/contract.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py +git commit -m "$(cat <<'EOF' +Add the shared CFDataset contract test + +One test body, subclassed per implementation, as the spec's testing +section requires. PR 4 supplies three fixtures and inherits the lot. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 6: `CFVariable.attributes`, and `__getattr__` rewired onto it + +The pivot of the PR. `cf_data` is still a netCDF4 variable at the end of this +task — only where CF attributes are *read from* changes. Tasks 8, 9 and 10 +then rewrite the consumers against `.attributes`, and Task 11 swaps the type +underneath them. + +Spec section 4.3, and its warning that this is "the sharpest edge in PR 2". + +**Files:** +- Modify: `lib/iris/fileformats/cf/_variables.py:97-250` +- Modify: `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py:137-180` +- Modify: `lib/iris/tests/unit/fileformats/cf/identify_mixins.py:21-35` +- Test: `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py` + +**Interfaces:** +- Consumes: `TrackedAttributes` from Task 1. +- Produces, on every `CFVariable` subclass: + - `attributes: TrackedAttributes` — the CF attributes, and the record of + which have been read. Tasks 8, 9, 10 and 12 all read through it. + - typed properties `dimensions: tuple`, `shape: tuple`, `ndim: int`, + `dtype`, `size: int`, all forwarding to `cf_data`. + - `__getattr__(name)` resolving against `attributes`, falling back to + `_deprecated_netcdf_member(name)`, and **not** caching onto the instance. + - `cf_attrs()`, `cf_attrs_ignored()`, `cf_attrs_used()`, + `cf_attrs_unused()`, `cf_attrs_reset()` — same signatures and same + `tuple[tuple[str, Any], ...]` returns as before, now computed from the + mapping. +- Removes: `CFVariable._nc_attrs` and `CFVariable._cf_attrs`. Nothing outside + `_variables.py` and its own unit test reads either — confirmed by + `grep -rn "_nc_attrs\|_cf_attrs" lib/`. + +- [ ] **Step 1: Rewrite PR 1's two caching tests to describe the new contract** + +In `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py`, replace +`TestAttributeAccess.test_cached` (line 138) and +`test_getattr_non_ncattr_value_is_cached_but_not_marked_used` (line 166) — +the only two tests in the PR 1 suite that name the instance cache, and so the +only two this PR is entitled to change (finding F7): + +```python +class TestAttributeAccess: + def test_reads_are_not_cached_on_the_instance(self, nc_var): + # The setattr cache is gone. It made a re-read after cf_attrs_reset() + # invisible to attribute tracking, which decided which attributes + # reached the cube - see TestReadAfterReset below. + cf_var = CFVariableSub("foo", nc_var) + + assert "coordinates" not in cf_var.__dict__ + assert cf_var.coordinates == "x y" + assert "coordinates" not in cf_var.__dict__ + assert cf_var.coordinates == "x y" + assert "coordinates" not in cf_var.__dict__ + + def test_attributes_are_read_from_the_file_once(self, nc_var): + # Materialised at construction, so repeated reads cost nothing and + # cf_attrs_unused() does not have to go back to the file to answer. + cf_var = CFVariableSub("foo", nc_var) + assert nc_var.ncattrs.call_count == 1 + + _ = cf_var.coordinates + _ = cf_var.standard_name + assert nc_var.ncattrs.call_count == 1 + assert nc_var.getncattr.call_count == 3 + + def test_getattr_of_a_non_attribute_reaches_the_variable(self, nc_var): + # The one-cycle compatibility route: a netCDF4 member that is not a + # CF attribute still resolves, and is still not marked as used. + nc_var.not_an_ncattr = 42 + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.not_an_ncattr == 42 + assert "not_an_ncattr" not in cf_var.__dict__ + assert "not_an_ncattr" not in dict(cf_var.cf_attrs()) + assert "not_an_ncattr" not in dict(cf_var.cf_attrs_used()) + + def test_getattr_of_nothing_at_all_raises_attribute_error(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + # A MagicMock invents any member, so ask a real object instead. + cf_var.cf_data = object() + with pytest.raises(AttributeError, match="nonesuch"): + cf_var.nonesuch +``` + +- [ ] **Step 2: Add the tests for the mapping itself and the typed properties** + +Append to `test_CFVariable.py`: + +```python +class TestAttributesMapping: + def test_attributes_contents(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + assert dict(cf_var.attributes) == { + "coordinates": "x y", + "standard_name": "air_temperature", + "_FillValue": -999, + } + + def test_getattr_and_mapping_are_the_same_read(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + _ = cf_var.coordinates + assert cf_var.attributes.read == frozenset(["_FillValue", "coordinates"]) + + cf_var.cf_attrs_reset() + _ = cf_var.attributes["coordinates"] + assert cf_var.attributes.read == frozenset(["_FillValue", "coordinates"]) + + def test_untracked_read_is_not_recorded(self, nc_var): + # What helpers.py's flag-attribute probe needs: a look that does not + # count, so the attribute still reaches the cube. + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.attributes.untracked["coordinates"] == "x y" + assert "coordinates" in dict(cf_var.cf_attrs_unused()) + + +class TestTypedProperties: + def test_dimensions_is_a_tuple(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + assert cf_var.dimensions == ("time", "lat") + + def test_shape_ndim_dtype_size(self, mocker, nc_var): + nc_var.shape = (3, 4) + nc_var.dtype = np.dtype("f4") + nc_var.size = 12 + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.shape == (3, 4) + assert cf_var.ndim == 2 + assert cf_var.dtype == np.dtype("f4") + assert cf_var.size == 12 + + def test_a_file_attribute_does_not_reach_the_typed_properties(self, nc_var): + # The properties are looked up before __getattr__ runs, so a file + # attribute called "shape" cannot displace the variable's shape. + nc_var.ncattrs.return_value = ["shape"] + nc_var.getncattr.side_effect = {"shape": "not a shape"}.__getitem__ + nc_var.shape = (2,) + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.shape == (2,) + assert cf_var.attributes["shape"] == "not a shape" +``` + +Add `import numpy as np` at the top of the file. + +- [ ] **Step 3: Give the identify() stub a `getncattr`** + +`CFVariable.__init__` now materialises attribute *values*, not just names, so +every stub standing in for a netCDF variable needs the method that fetches +them. In `lib/iris/tests/unit/fileformats/cf/identify_mixins.py`, add to +`_NetCDFVar`: + +```python + def getncattr(self, name): + return getattr(self, name) +``` + +`test_CFCoordinateVariable.py`'s stub returns `[]` from `ncattrs()`, so it +needs nothing. `test_CFReader.py` uses `MagicMock`, which supplies +`getncattr` already. + +- [ ] **Step 4: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/test_CFVariable.py` +Expected: FAIL — `AttributeError` on `cf_var.attributes` (resolved through the +old `__getattr__` to the mock, so it will report as a wrong *value* rather +than a missing member: `assert dict() == {...}` raising +`TypeError`), and `assert "coordinates" not in cf_var.__dict__` failing on +the second read. + +- [ ] **Step 5: Rewire `CFVariable`** + +In `lib/iris/fileformats/cf/_variables.py`, add below `_CF_ATTRS_IGNORE`: + +```python +# Names __getattr__ must never resolve through self.attributes, because +# reading self.attributes is how it resolves anything at all. Deliberately +# not "every underscore-prefixed name": loader.py and saver.py both probe +# for "_data_array", and that probe has to reach through. +_GETATTR_RECURSION_GUARD = frozenset(["attributes", "cf_data"]) + + +def _attributes_source(data): + """Return a mapping of the CF attributes of a variable's backing store.""" + # PR 2 transitional. Task 11 deletes this: once CFReader supplies + # CFDatasetVariables, __init__ reads data.attributes directly. + if isinstance(data, CFDatasetVariable): + return data.attributes + return {name: data.getncattr(name) for name in data.ncattrs()} +``` + +`isinstance`, not `hasattr(data, "attributes")`. Half the CF test suite hands +this function a `MagicMock`, and a `MagicMock` has every attribute you ask it +for — including `attributes`, which would send it down the wrong branch and +put a mock where the mapping belongs. + +Add `CFDatasetVariable` to the imports too: + +```python +from iris.fileformats.cf.dataset import CFDatasetVariable, TrackedAttributes +``` + +`cf/dataset.py` imports nothing from `iris.fileformats.netcdf` (Task 1, +Step 10), so this does not breach PR 1's +`test_only_the_reader_is_coupled_to_netcdf`. + +Replace `CFVariable.__init__`: + +```python + def __init__(self, name, data): + self.cf_name = name + """NetCDF variable name.""" + + self.cf_data = data + """The variable's storage: a CFDatasetVariable.""" + + self.attributes = TrackedAttributes( + _attributes_source(data), ignored=_CF_ATTRS_IGNORE + ) + """The variable's CF attributes, and a record of which have been read. + + The only place CF attributes live. ``__getattr__`` forwards here, so + ``cf_var.units`` and ``cf_var.attributes["units"]`` are the same read + and count once. What was read decides what survives onto the loaded + cube - see :func:`iris.fileformats.netcdf.loader._add_unused_attributes`. + """ + + """File source of the NetCDF content.""" + try: + self.filename = data.group().filepath() + except AttributeError: + self.filename = "" + + self.cf_group = None + """Collection of CF-netCDF variables associated with this variable.""" + + self.cf_terms_by_root = {} + """CF-netCDF formula terms that his variable participates in.""" + + self._to_be_promoted = False +``` + +The trailing `self.cf_attrs_reset()` goes: `TrackedAttributes` seeds itself +from `ignored` at construction, which is what that call was for. + +Replace `__getattr__` (line 203) with: + +```python + @property + def dimensions(self) -> tuple: + """The names of the dimensions this variable spans, in order.""" + return tuple(self.cf_data.dimensions) + + @property + def shape(self) -> tuple: + """The variable's shape.""" + return self.cf_data.shape + + @property + def ndim(self) -> int: + """The number of dimensions this variable spans.""" + return len(self.shape) + + @property + def dtype(self): + """The variable's stored data type.""" + return self.cf_data.dtype + + @property + def size(self) -> int: + """The total number of elements in the variable.""" + return self.cf_data.size + + def __getattr__(self, name): + """Return the named CF attribute, as read from the file. + + The open-ended half of this class. CF attribute names are data read + from a file, not API, so they cannot be declared - which is what + justifies ``__getattr__`` here at all. It resolves only against + :attr:`attributes`; everything structural is a declared property + above. Reading through it records the attribute as used, exactly as + ``cf_var.attributes[name]`` does. + + """ + if name.startswith("__") or name in _GETATTR_RECURSION_GUARD: + # Dunder probes - copy, pickle, numpy protocols - must not be + # answered from file data, and the two members this method reads + # must not be resolved by this method. + raise AttributeError(name) + + try: + return self.attributes[name] + except KeyError: + pass + return self._deprecated_netcdf_member(name) + + def _deprecated_netcdf_member(self, name): + """Return a netCDF4 member of the backing variable, as this class used to. + + The one-cycle compatibility route for code that reached netCDF4 API + through a CFVariable. Records nothing: this is not a CF attribute. + + """ + # PR 2 transitional. Task 11 replaces the body with + # value = self.cf_data.deprecated_netcdf_member(name) + # and the deprecation warning, once no Iris code takes this path. + return getattr(self.cf_data, name) +``` + +Replace the five `cf_attrs_*` helpers (lines 226-250): + +```python + def cf_attrs(self): + """Return a list of all attribute name and value pairs of the CF-netCDF variable.""" + return tuple( + (name, self.attributes.untracked[name]) for name in sorted(self.attributes) + ) + + def cf_attrs_ignored(self): + """Return a list of all ignored attribute name and value pairs of the CF-netCDF variable.""" + names = set(self.attributes) & _CF_ATTRS_IGNORE + return tuple( + (name, self.attributes.untracked[name]) for name in sorted(names) + ) + + def cf_attrs_used(self): + """Return a list of all accessed attribute name and value pairs of the CF-netCDF variable.""" + return tuple( + (name, self.attributes.untracked[name]) + for name in sorted(self.attributes.read) + ) + + def cf_attrs_unused(self): + """Return a list of all non-accessed attribute name and value pairs of the CF-netCDF variable.""" + return tuple( + (name, self.attributes.untracked[name]) + for name in sorted(self.attributes.unread) + ) + + def cf_attrs_reset(self): + """Reset the history of accessed attribute names of the CF-netCDF variable.""" + self.attributes.reset() +``` + +Every one of these reads through `.untracked`: a report on what was read must +not itself count as reading. + +`__getitem__`, `__len__`, `__repr__`, `__eq__`, `__ne__`, `__hash__` and +`spans` are unchanged. `spans` already reads `self.dimensions`, which is now +the declared property rather than a `__getattr__` hop. + +- [ ] **Step 6: Run the tests to verify they pass** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/` +Expected: PASS — 0 failed, 1 skipped (§6.2; the skip is `test_CFReader.py`'s +FUTURE-context case and is expected). The passed count will be higher than +§6.2's 306, by the tests Tasks 1–5 added under `cf/dataset/`. This includes +every `test_CF*Variable.py` `identify()` test, which is the broadest check +that the rewiring did not change what the CF layer sees. + +- [ ] **Step 7: Run the consumers, which still take the fallback** + +Nothing outside `cf/` has been touched yet, so every `getattr(cf_var, ...)` +in `_nc_load_rules/`, `netcdf/loader.py` and `netcdf/ugrid_load.py` must still +work — through `attributes` for CF attributes, through +`_deprecated_netcdf_member` for the rest. + +Run: `pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/` +Expected: `0 failed, 26 skipped`, plus F14's ten `test_coord_systems.py` +errors and nothing else (§6.2). The skip count must be *identical* to the +baseline, not merely small. + +If `lib/iris/tests/integration/test_cf.py::test_variable_attribute_touch_pass_0` +fails, read it before changing anything: it pins the tracking semantics on a +real file and is a PR 1 test that must pass untouched. + +- [ ] **Step 8: Commit the rewiring** + +```bash +pre-commit run --files lib/iris/fileformats/cf/_variables.py \ + lib/iris/tests/unit/fileformats/cf/test_CFVariable.py \ + lib/iris/tests/unit/fileformats/cf/identify_mixins.py +git add lib/iris/fileformats/cf/_variables.py \ + lib/iris/tests/unit/fileformats/cf/test_CFVariable.py \ + lib/iris/tests/unit/fileformats/cf/identify_mixins.py +git commit -m "$(cat <<'EOF' +Resolve CFVariable attributes against a declared mapping + +CFVariable.__getattr__ now reads self.attributes - a TrackedAttributes - +instead of reaching for arbitrary members of the netCDF4 variable, and no +longer caches what it finds onto the instance. dimensions, shape, ndim, +dtype and size become declared properties, so structure and file data are +no longer reached for through the same syntax. + +The instance cache had to go: it made a re-read after cf_attrs_reset() +invisible to attribute tracking, and tracking is what decides which file +attributes survive onto a loaded cube. + +cf_data is still a netCDF4 variable; that swap is later in this PR. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 9: Review Focus 1 — write the failing test for a shadowing attribute** + +A file is free to carry an attribute called `filename`, `cf_name`, `spans`, +`attributes` or `cf_data`. `__getattr__` runs only when normal lookup fails, +so for those five names it never runs and the class member wins. The spec +calls this a known limitation; what it must not be is a *silent* one, and +what must never happen is the attribute vanishing from the cube because +nothing recorded whether it was read. + +Append to `test_CFVariable.py`: + +```python +#: CFVariable members that a CF attribute of the same name cannot displace. +SHADOWED_NAMES = ["filename", "cf_name", "spans", "attributes", "cf_data"] + + +class TestShadowedAttributeNames: + """Review Focus 1. Spec section 4.3's known limitation, pinned.""" + + @pytest.fixture + def shadowing(self, nc_var): + nc_var.ncattrs.return_value = SHADOWED_NAMES + ["units"] + nc_var.getncattr.side_effect = ( + {name: f"file value of {name}" for name in SHADOWED_NAMES} | {"units": "K"} + ).__getitem__ + return CFVariableSub("foo", nc_var) + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_the_class_member_wins(self, shadowing, name): + assert getattr(shadowing, name) != f"file value of {name}" + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_the_file_value_is_still_reachable(self, shadowing, name): + assert shadowing.attributes[name] == f"file value of {name}" + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_reading_it_through_the_mapping_marks_it_used(self, shadowing, name): + assert name in dict(shadowing.cf_attrs_unused()) + _ = shadowing.attributes[name] + assert name in dict(shadowing.cf_attrs_used()) + assert name not in dict(shadowing.cf_attrs_unused()) + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_it_is_listed_among_the_variables_attributes(self, shadowing, name): + assert name in dict(shadowing.cf_attrs()) + + def test_a_shadowing_name_does_not_disturb_its_neighbours(self, shadowing): + assert shadowing.units == "K" + assert dict(shadowing.cf_attrs_used())["units"] == "K" +``` + +- [ ] **Step 10: Run them** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/test_CFVariable.py -k Shadowed -v` +Expected: PASS on the implementation from Step 5. If +`test_the_file_value_is_still_reachable` fails, `__init__` is dropping +attributes whose names collide with members — which is the failure this whole +class exists to catch. + +- [ ] **Step 11: Review Focus 2 — `getattr` with a default, and `hasattr`** + +`_nc_load_rules/helpers.py` alone makes 63 of these calls, almost all +`getattr(cf_var, SOME_CF_ATTR, None)`. They rely on `__getattr__` raising +`AttributeError` and nothing else: a `KeyError` escaping from the mapping, or +a `TypeError` from an unhashable name, would not be caught by the default and +would abort the load. And `hasattr` must keep marking the attribute used, or +an attribute Iris consulted would go on to be copied onto the cube as well. + +Append to `test_CFVariable.py`: + +```python +class TestGetattrAndHasattr: + """Review Focus 2. The two call shapes the loading rules actually use.""" + + def test_getattr_with_a_default_finds_the_attribute(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + assert getattr(cf_var, "coordinates", None) == "x y" + + def test_getattr_with_a_default_returns_the_default(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + assert getattr(cf_var, "nonesuch", None) is None + assert getattr(cf_var, "nonesuch", "fallback") == "fallback" + + def test_a_missing_attribute_raises_attribute_error_not_key_error(self, nc_var): + # getattr(..., default) only swallows AttributeError. A KeyError from + # the mapping would escape and abort the load. + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + with pytest.raises(AttributeError): + cf_var.nonesuch + + def test_getattr_of_a_default_does_not_record_a_read(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + _ = getattr(cf_var, "nonesuch", None) + assert cf_var.cf_attrs_used() == (("_FillValue", -999),) + + def test_hasattr_true_marks_the_attribute_used(self, nc_var): + # Parity with the old behaviour: hasattr went through __getattr__, + # which added the name to the used set. + cf_var = CFVariableSub("foo", nc_var) + assert hasattr(cf_var, "coordinates") + assert "coordinates" in dict(cf_var.cf_attrs_used()) + + def test_hasattr_false_marks_nothing(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + assert not hasattr(cf_var, "nonesuch") + assert cf_var.cf_attrs_used() == (("_FillValue", -999),) + + def test_dunder_probes_do_not_reach_the_file(self, nc_var): + # copy, pickle and numpy all probe for dunders. Answering one from + # file data would make a CFVariable behave as whatever the file says. + nc_var.ncattrs.return_value = ["__array__"] + nc_var.getncattr.side_effect = {"__array__": "nonsense"}.__getitem__ + cf_var = CFVariableSub("foo", nc_var) + + assert not hasattr(cf_var, "__array_interface__") + assert cf_var.attributes.untracked["__array__"] == "nonsense" + + @pytest.mark.parametrize("name", ["attributes", "cf_data"]) + def test_the_recursion_guard_holds_before_init_completes(self, name): + # An instance whose __init__ never ran - what copy and pickle build - + # must raise, not recurse until the stack is gone. + cf_var = CFVariableSub.__new__(CFVariableSub) + with pytest.raises(AttributeError, match=name): + getattr(cf_var, name) +``` + +- [ ] **Step 12: Run them** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/test_CFVariable.py -k "GetattrAndHasattr" -v` +Expected: PASS. A `RecursionError` on the last two means +`_GETATTR_RECURSION_GUARD` is missing a name. + +- [ ] **Step 13: Review Focus 3 — a read repeated after `cf_attrs_reset()`** + +The behaviour the removed cache was hiding. `CFReader` resets every variable's +touch history once it has finished parsing the file's structure, so any +attribute the rules go on to read is one the rules used. With the cache, a +name read during parsing and read again by the rules was served from +`__dict__`, never re-recorded, and so was copied onto the cube as an "unused" +attribute as well as being acted on. Removing the cache fixes that; this +pins it so it cannot come back. + +Append to `test_CFVariable.py`: + +```python +class TestReadAfterReset: + """Review Focus 3. The cache removal's whole point, at unit scale. + + Task 11, Step 1 pins the same behaviour end to end, on a real file. + """ + + def test_a_re_read_after_reset_counts_again(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + # As CFReader does: read while parsing structure, then reset. + _ = cf_var.coordinates + assert "coordinates" in dict(cf_var.cf_attrs_used()) + cf_var.cf_attrs_reset() + assert "coordinates" in dict(cf_var.cf_attrs_unused()) + + # As the loading rules then do: read it again. + _ = cf_var.coordinates + assert "coordinates" in dict(cf_var.cf_attrs_used()) + assert "coordinates" not in dict(cf_var.cf_attrs_unused()) + + def test_the_same_holds_through_hasattr(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + _ = cf_var.coordinates + cf_var.cf_attrs_reset() + + assert hasattr(cf_var, "coordinates") + assert "coordinates" in dict(cf_var.cf_attrs_used()) + + def test_reset_restores_the_ignored_names_only(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + _ = cf_var.coordinates + _ = cf_var.standard_name + cf_var.cf_attrs_reset() + + assert cf_var.cf_attrs_used() == (("_FillValue", -999),) + assert cf_var.cf_attrs_unused() == ( + ("coordinates", "x y"), + ("standard_name", "air_temperature"), + ) + + def test_values_are_unchanged_by_any_of_this(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + for _ in range(3): + assert cf_var.coordinates == "x y" + cf_var.cf_attrs_reset() +``` + +- [ ] **Step 14: Run them** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/test_CFVariable.py -k ReadAfterReset -v` +Expected: PASS. + +- [ ] **Step 15: Green check and commit the Review Focus tests** + +```bash +pre-commit run --files lib/iris/tests/unit/fileformats/cf/test_CFVariable.py +pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/ +git add lib/iris/tests/unit/fileformats/cf/test_CFVariable.py +git commit -m "$(cat <<'EOF' +Pin the sharp edges of the CFVariable attribute rewiring + +Three classes of input the rewiring could plausibly break: a file +attribute whose name collides with a CFVariable member; getattr with a +default and hasattr, which the loading rules use 86 times between them +and which depend on AttributeError and nothing else; and an attribute +read again after cf_attrs_reset(), which is what removing the instance +cache actually changes. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 7: One definition of "scalar" across the four `spans` methods + +Spec section 4.4 names this the programme's only `bugfix`: `CFVariable.spans` +recognises NCZarr's `_scalar_` pseudo-dimension and the three subclasses that +override `spans` do not. **Finding F9 corrects that** — measured against the +real classes, the three overrides already return `True` for a source whose +dimensions are exactly `("_scalar_",)`, because they open with +`if self.dimensions:` and then compare `set(source[:-1])`, which for a +one-element tuple is the empty set, a subset of everything. The gap is real +in the source and unreachable in behaviour. + +So this task does not fix an outcome. It replaces four separate spellings of +"is this variable scalar" with one, which matters from PR 4 on: `ZarrDataset` +derives `dimensions` from `_ARRAY_DIMENSIONS` or NCZarr metadata (spec +section 4.4), and the question "does `_scalar_` count as scalar here?" will +be asked again by a backend that has never been through netCDF's rules. + +**Because the behaviour is already correct, this task inverts the usual +cycle**: the tests are written first and expected to *pass*, recording what +the code does today; the refactor must then leave them passing. Do not +"correct" a step that says PASS where a step usually says FAIL. + +**Files:** +- Modify: `lib/iris/fileformats/cf/_variables.py` — `CFVariable.spans`, + `CFBoundaryVariable.spans:420`, `CFClimatologyVariable.spans:496`, + `CFLabelVariable.spans:819` +- Test: `lib/iris/tests/unit/fileformats/cf/identify_mixins.py` — `SpansMixin`, + which `test_CFBoundaryVariable.py`, `test_CFClimatologyVariable.py` and + `test_CFLabelVariable.py` all subclass, so one test there covers all three +- Test: `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py` — `TestSpans` + +**Interfaces:** +- Produces: `CFVariable._is_scalar() -> bool` — true for a variable with no + dimensions, and for one whose only dimension is + `_NCZARR_SCALAR_DIMENSION`. The single definition of scalar in the CF + layer; PR 4's Zarr work reads it rather than restating the rule. + +- [ ] **Step 1: Add the characterisation tests to the shared mixin** + +In `lib/iris/tests/unit/fileformats/cf/identify_mixins.py`, add to +`SpansMixin`, after `test_empty_dimensions_spans`: + +```python + def test_nczarr_scalar_dimension_spans(self): + """An NCZarr scalar source variable always spans the target. + + NCZarr spells a zero-dimensional variable as one dimension named + _scalar_, so this is the same case as test_empty_dimensions_spans + arriving from a different writer. + """ + cf_source = self._make_cf_var("source_var", (_NCZARR_SCALAR_DIMENSION,)) + cf_target = self._make_cf_var("target_var", ("x", "y")) + assert cf_source.spans(cf_target) + + def test_nczarr_scalar_target_is_spanned_by_a_scalar(self): + cf_source = self._make_cf_var("source_var", (_NCZARR_SCALAR_DIMENSION,)) + cf_target = self._make_cf_var("target_var", (_NCZARR_SCALAR_DIMENSION,)) + assert cf_source.spans(cf_target) + + def test_nczarr_scalar_is_not_a_dimension_name_to_match_on(self): + """_scalar_ marks a scalar; it is not a dimension two variables share. + + A source that is not scalar has to justify itself dimension by + dimension, whether or not _scalar_ is among its names. + """ + cf_source = self._make_cf_var("source_var", ("a", "b", "extra")) + cf_target = self._make_cf_var("target_var", (_NCZARR_SCALAR_DIMENSION,)) + assert not cf_source.spans(cf_target) +``` + +Add the import at the top of the file: + +```python +from iris.fileformats.cf._variables import _NCZARR_SCALAR_DIMENSION +``` + +- [ ] **Step 2: Add the matching tests for the base class** + +In `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py`, add to +`TestSpans`: + +```python + def test_scalar_dimension_with_others_is_not_scalar(self, mocker, nc_var): + # _scalar_ means "this variable is scalar", which a variable with a + # second dimension is not. It is not a wildcard. + nc_var.dimensions = (_variables._NCZARR_SCALAR_DIMENSION, "bnds") + cf_var = CFVariableSub("not_scalar", nc_var) + + other = mocker.MagicMock() + other.dimensions = ("time",) + + assert not cf_var.spans(other) + + def test_no_dimensions_always_spans(self, mocker, nc_var): + nc_var.dimensions = () + cf_var = CFVariableSub("scalar", nc_var) + + other = mocker.MagicMock() + other.dimensions = ("time",) + + assert cf_var.spans(other) +``` + +- [ ] **Step 3: Run them, and expect PASS** + +Run: +```bash +pytest lib/iris/tests/unit/fileformats/cf/test_CFVariable.py \ + lib/iris/tests/unit/fileformats/cf/test_CFBoundaryVariable.py \ + lib/iris/tests/unit/fileformats/cf/test_CFClimatologyVariable.py \ + lib/iris/tests/unit/fileformats/cf/test_CFLabelVariable.py -k "span or Span" -v +``` +Expected: PASS, all of them, on unmodified code. That result **is** the +finding — record the exact output in the pull request body, because it is the +evidence for F9 and for softening the merge-back's `bugfix` wording. + +If any of them fails, F9 is wrong and the spec was right: stop, and say so in +the pull request. The refactor below is then a genuine fix, and the merge-back +owes the `bugfix` fragment as originally specified. + +- [ ] **Step 4: Commit the characterisation tests, before changing anything** + +```bash +pre-commit run --files lib/iris/tests/unit/fileformats/cf/identify_mixins.py \ + lib/iris/tests/unit/fileformats/cf/test_CFVariable.py +git add lib/iris/tests/unit/fileformats/cf/identify_mixins.py \ + lib/iris/tests/unit/fileformats/cf/test_CFVariable.py +git commit -m "$(cat <<'EOF' +Pin how spans() treats the NCZarr _scalar_ dimension + +Committed before the code changes, so the record shows these pass +against unmodified spans() methods. + +The design spec calls the missing _NCZARR_SCALAR_DIMENSION check in the +three overriding spans() methods a latent bug. It is latent in the +strong sense: unreachable. Each override opens with "if self.dimensions" +and then tests set(source[:-1]), which for the one-element tuple NCZarr +produces is the empty set - a subset of every target. The overrides +already answer True. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 5: Give `CFVariable` one definition of scalar** + +In `lib/iris/fileformats/cf/_variables.py`, add to `CFVariable`, immediately +above `spans`: + +```python + def _is_scalar(self) -> bool: + """Whether this variable is scalar, in either spelling of scalar. + + NetCDF gives a zero-dimensional variable no dimensions at all. + NCZarr gives it one, named ``_scalar_``. The two mean the same + thing, and every ``spans`` implementation has to treat them alike. + + """ + return not self.dimensions or self.dimensions == (_NCZARR_SCALAR_DIMENSION,) +``` + +Rewrite the body of `CFVariable.spans` (keeping its docstring): + +```python + # Scalar variables always span the target variable. + if self._is_scalar(): + return True + + result = set(self.dimensions).issubset(cf_variable.dimensions) + return result +``` + +This drops the local `dimensions = tuple(self.dimensions)`: Task 6 made +`CFVariable.dimensions` a property that already returns a tuple. + +- [ ] **Step 6: Use it in the three overrides** + +In each of `CFBoundaryVariable.spans`, `CFClimatologyVariable.spans` and +`CFLabelVariable.spans`, replace `if self.dimensions:` with +`if not self._is_scalar():`. Nothing else in those three bodies changes — +including the comment above each subset test, which names a different extent +dimension in each class and must keep doing so. The result in +`CFBoundaryVariable`: + +```python + # Scalar variables always span the target variable. + result = True + if not self._is_scalar(): + source = self.dimensions + target = cf_variable.dimensions + # Ignore the bounds extent dimension. + result = set(source[:-1]).issubset(target) or set(source[1:]).issubset( + target + ) + return result +``` + +- [ ] **Step 7: Run the tests again, and expect the same PASS** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/` +Expected: PASS, 0 failed, 1 skipped (the §6.2 FUTURE-context skip) — the +same set of tests, the same results, over +consolidated code. `test_CFReader.py::test_nczarr_scalar_grid_mapping_spans_data_var` +is the one existing test that exercises `_scalar_` through the reader; it must +still pass. + +- [ ] **Step 8: Run the loading path** + +Run: `pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/` +Expected: `0 failed, 26 skipped`, plus F14's ten `test_coord_systems.py` +errors and nothing else (§6.2). `spans` decides which coordinates attach to +which cube, so a mistake here shows up as missing or duplicated coordinates +rather than as an error. + +- [ ] **Step 9: Commit** + +```bash +pre-commit run --files lib/iris/fileformats/cf/_variables.py +git add lib/iris/fileformats/cf/_variables.py +git commit -m "$(cat <<'EOF' +Ask _is_scalar() rather than restate the rule four times + +CFVariable.spans and its three overrides each decided for themselves +what a scalar variable is, and only one of the four knew about NCZarr's +_scalar_ pseudo-dimension. The behaviour is unchanged - see the tests +committed just before this - but the question is now asked in one place. + +That matters from PR 4: ZarrDataset derives dimension names from the +store rather than from netCDF's rules, and will ask the same question of +data that has never been through them. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 8: The loading rules read `.attributes` + +66 sites, 63 of them the same two lines apart. Mechanical, but it is the task +that earns the `__getattr__` exemption in `lib/iris/AGENTS.md`: *"It must +forward to a declared `Mapping`, that `Mapping` must be the path library code +takes."* Until this task, the second half of that sentence is false. + +Nothing here changes behaviour. `getattr(cf_var, CF_ATTR_UNITS, None)` already +resolves through `.attributes` after Task 6; this task makes the read say so. + +**Files:** +- Modify: `lib/iris/fileformats/_nc_load_rules/helpers.py` — 63 sites between + line 567 and line 2077 +- Modify: `lib/iris/fileformats/_nc_load_rules/actions.py:179,560,626` +- Test: `lib/iris/tests/unit/fileformats/nc_load_rules/helpers/` — the existing + modules, unchanged; they are the regression net for this task + +**Interfaces:** +- Consumes: `CFVariable.attributes` from Task 6. +- Produces: nothing new. This task removes the loading rules' dependence on + `CFVariable.__getattr__`, which is what lets Task 11 make the netCDF + fallback warn. + +**The rule, applied 63 times:** + +| Before | After | +|---|---| +| `getattr(cf_var, CF_ATTR_UNITS, None)` | `cf_var.attributes.get(CF_ATTR_UNITS)` | +| `getattr(cf_grid_var, CF_ATTR_GRID_NORTH_POLE_LAT, 90.0)` | `cf_grid_var.attributes.get(CF_ATTR_GRID_NORTH_POLE_LAT, 90.0)` | +| `getattr(cf_var, CF_ATTR_AXIS, "")` | `cf_var.attributes.get(CF_ATTR_AXIS, "")` | +| `getattr(cf_grid_var, CF_ATTR_GRID_MAPPING_NAME)` | `cf_grid_var.attributes[CF_ATTR_GRID_MAPPING_NAME]` | + +A default of `None` is `Mapping.get`'s own default, so drop it; any other +default is passed through. The constants keep their names — they are the +greppable literals `lib/iris/AGENTS.md` asks for, and the point of this task +is the object being asked, not the key. + +- [ ] **Step 1: Rewrite the 63 plain sites in `helpers.py`** + +Work top to bottom, applying the table above. The sites, by line number in the +unmodified file: + +``` +567 616 625 745 774 775 776 784 786 789 832 833 840 863 864 +903 904 905 934 935 961 966 967 990 991 992 1020 1021 1045 1046 +1047 1075 1076 1104 1105 1106 1126 1136 1137 1149 1169 1220 1231 1232 1267 +1268 1836 1844 1850 1855 1865 1905 1915 1929 1931 1932 1948 1964 2012 2017 +2046 2077 +``` + +Line 1905 and line 1915 each hold two `getattr` calls in one expression: + +```python + attr_name = getattr(cf_var, CF_ATTR_STD_NAME, None) or getattr( + cf_var, CF_ATTR_LONG_NAME, None + ) +``` + +becomes + +```python + attr_name = cf_var.attributes.get(CF_ATTR_STD_NAME) or cf_var.attributes.get( + CF_ATTR_LONG_NAME + ) +``` + +Line 1149 is the only site with no default. It reads `grid_mapping_name` from +a variable that was identified as a grid mapping *because* it carries that +attribute, so it cannot miss; make it an indexing read and let a miss be a +`KeyError`: + +```python + # Handle the alternative form noted in CF: rotated mercator. + grid_mapping_name = cf_grid_var.attributes[CF_ATTR_GRID_MAPPING_NAME] +``` + +`_add_or_capture_load_problem` catches `Exception`, so a `KeyError` here is +recorded exactly as the `AttributeError` was. + +- [ ] **Step 2: Rewrite `helpers.py:567`, the computed-name site** + +`attr_key` is a variable, not a literal — `lib/iris/AGENTS.md` bans `getattr` +string dispatch, and the mapping is the sanctioned form: + +```python + if attr_key is not None: + captured_attr = None + with contextlib.suppress(KeyError): + captured_attr = cf_var.attributes[attr_key] + captured = {attr_key: captured_attr} +``` + +The suppressed exception changes with the call. Do not leave +`suppress(AttributeError)` in place "just in case": it would silently stop +suppressing, and this block runs while already handling a load failure, so a +second exception escaping it loses the first. + +- [ ] **Step 3: Rewrite `helpers.py:1213`, the deliberately untracked probe** + +This one asks whether a flag attribute exists *without* marking it read, so +that it still reaches the cube. Today it says so by reaching past the +`CFVariable` to `cf_data`, which Task 11 would break. + +```python + if any( + # Deliberately untracked: asking whether these exist must not count as + # using them, or they would stop being copied onto the cube. + name in cf_var.attributes.untracked + for name in ("flag_values", "flag_masks", "flag_meanings") + ): + attr_units = cf_units._NO_UNIT_STRING +``` + +- [ ] **Step 4: Rewrite the three sites in `actions.py`** + +Line 179 and line 626 take the plain rule: + +```python + grid_mapping_type = cf_var.attributes.get(hh.CF_ATTR_GRID_MAPPING_NAME) +``` + +```python + # cf_var.standard_name is a formula type (or we should never get here). + formula_type = cf_var.attributes.get("standard_name") +``` + +Line 560 is a second computed-name site — `match_name` comes from +`handler.netcdf_names`: + +```python + for match_name in handler.netcdf_names: + match_value = var.attributes.get(match_name) +``` + +- [ ] **Step 5: Check no site was missed** + +Run: +```bash +grep -n "getattr(cf_\|hasattr(cf_\|getattr(var,\|hasattr(var," \ + lib/iris/fileformats/_nc_load_rules/helpers.py \ + lib/iris/fileformats/_nc_load_rules/actions.py +``` +Expected: no output. A match is a site still resolving through +`CFVariable.__getattr__`, which Task 11 will make warn. + +- [ ] **Step 6: Run the loading rules' own tests** + +Run: `pytest lib/iris/tests/unit/fileformats/nc_load_rules/` +Expected: PASS, 0 failed, 0 skipped. These tests were written against the +`getattr` form and are not modified by this task — which is exactly what makes +them evidence. + +- [ ] **Step 7: Run everything that loads a file** + +Run: +```bash +pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/ \ + lib/iris/tests/test_netcdf.py +``` +Expected: `0 failed`, the §6.2 skip count for this selection, and F14's ten +`test_coord_systems.py` errors — nothing else. + +Watch for cube attributes appearing or disappearing rather than for errors: +this task moves reads onto the tracking mapping, and a read that stopped +counting shows up as an extra attribute on a loaded cube, not as a traceback. +`lib/iris/tests/integration/test_netcdf__loadsaveattrs.py` is the densest +check of that and must be clean. + +- [ ] **Step 8: Commit** + +```bash +pre-commit run --files lib/iris/fileformats/_nc_load_rules/helpers.py \ + lib/iris/fileformats/_nc_load_rules/actions.py +git add lib/iris/fileformats/_nc_load_rules/helpers.py \ + lib/iris/fileformats/_nc_load_rules/actions.py +git commit -m "$(cat <<'EOF' +Read CF attributes from cf_var.attributes in the loading rules + +66 sites, 63 of them the same rewrite: getattr(cf_var, CF_ATTR_X, None) +becomes cf_var.attributes.get(CF_ATTR_X). Three are not: + +- helpers.py:567 and actions.py:560 look up a name held in a variable. + getattr string dispatch is banned by lib/iris/AGENTS.md; a mapping + lookup is the sanctioned form, and reads as one. +- helpers.py:1213 asks whether a flag attribute exists without marking it + used, so that it still reaches the cube. It said so by reaching past + the CFVariable to the netCDF4 object; it now says so with .untracked. + +No behaviour change: __getattr__ already resolved these through +.attributes. The rules now take that path openly, which is what lets +the netCDF fallback start warning later in this PR. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 9: `netcdf/loader.py` and `netcdf/ugrid_load.py` read `.attributes` + +The same rewrite as Task 8 in the two netCDF loading modules — **and only the +CF-attribute half of it**. Both modules also reach through `cf_var.cf_data` +for netCDF storage API: `_data_array`, `datatype`, `chunking()`, the proxy +class. Those cannot move until `cf_data` is a `CFDatasetVariable`, which is +Task 11. Mixing the two here would leave the tree red in between. + +A `hasattr` in this task matters more than in Task 8. `_add_unused_attributes` +copies onto the cube every attribute nothing read, so a probe that stops +counting as a read adds an attribute to loaded cubes. +`TrackedAttributes.__contains__` records a hit for exactly this reason, which +makes `X in cf_var.attributes` the faithful replacement for +`hasattr(cf_var, "X")`. + +**Files:** +- Modify: `lib/iris/fileformats/netcdf/loader.py:214,216,306,350,418,651-653,731` +- Modify: `lib/iris/fileformats/netcdf/ugrid_load.py:230,322,326,336,338,340,346,354,370,371,383,385,406,456` +- Test: `lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py` + — its doubles change, its assertions do not (Step 7) +- Test: the rest of `lib/iris/tests/unit/fileformats/netcdf/loader/` and + `lib/iris/tests/unit/fileformats/netcdf/test_ugrid_load.py` — unchanged + +**Interfaces:** +- Consumes: `CFVariable.attributes`, `CFVariable.dimensions` from Task 6. +- Produces: nothing new. + +**Deliberately left for Task 11**, because they are storage, not attributes: +`loader.py:242` `_data_array`, `:254` `datatype`, `:316` the +`EncodedVariable` proxy switch, `:321` the proxy construction, `:327` +`chunking()`. + +- [ ] **Step 1: `_get_actual_dtype` — loader.py:214-216** + +```python +def _get_actual_dtype(cf_var): + # Figure out what the eventual data type will be after any scale/offset + # transforms. + dummy_data = np.zeros(1, dtype=cf_var.dtype) + if "scale_factor" in cf_var.attributes: + dummy_data = cf_var.attributes["scale_factor"] * dummy_data + if "add_offset" in cf_var.attributes: + dummy_data = cf_var.attributes["add_offset"] + dummy_data + return dummy_data.dtype +``` + +`scale_factor` and `add_offset` are both in `_CF_ATTRS_IGNORE`, so they are +seeded as read and this cannot change what reaches the cube either way. + +- [ ] **Step 2: The fill value — loader.py:306** + +```python + fill_dtype = "S1" if cf_var.dtype is str else cf_var.dtype.str[1:] + fill_value = cf_var.attributes.get( + "_FillValue", _thread_safe_nc.default_fillvals[fill_dtype] + ) +``` + +- [ ] **Step 3: The chunking dimensions — loader.py:350** + +```python + dims = cf_var.dimensions +``` + +`CFVariable.dimensions` is a declared property as of Task 6 and returns the +same tuple `cf_var.cf_data.dimensions` did. + +- [ ] **Step 4: The dataless-cube marker — loader.py:418** + +```python + if Saver._DATALESS_ATTRNAME in cf_var.attributes: +``` + +- [ ] **Step 5: The constraint fast path — loader.py:641-655** + +Three things at once here: the computed name, the read, and a comment that +Task 6 made false. + +```python + def inner(cf_datavar): + match_any_constraint = False + for constraint in constraints: + match_this_constraint = True + for name in constraint._names: + expected = getattr(constraint, name) + if name != "STASH" and expected != "none": + if name == "var_name": + # Iris's name for it; not a file attribute at all. + actual = cf_datavar.cf_name + elif name in cf_datavar.attributes: + actual = cf_datavar.attributes[name] + else: + # Unlike NameConstraint, a variable that does not + # carry the attribute is not a mismatch here: the + # cube may still acquire the name later in the load. + continue + if actual != expected: + match_this_constraint = False + break + if match_this_constraint: + match_any_constraint = True + break + return match_any_constraint +``` + +Deleted with it: `# Fetch property : N.B. CFVariable caches the property +values`. Task 6 removed that cache, and `lib/iris/AGENTS.md` requires a +comment your change invalidates to be fixed or deleted. Also gone is +`attr_name = "cf_name" if name == "var_name" else name` — the computed name. + +`getattr(constraint, name)` stays. `constraint` is an +`iris._constraints.NameConstraint`, not a `CFVariable`; rewriting it is +outside this pull request. + +- [ ] **Step 6: The mesh reference — loader.py:731** + +```python + mesh_name = cf_var.attributes.get("mesh") +``` + +- [ ] **Step 7: Give the constraint callback's doubles real CF attributes** + +Step 5's rewrite is the first consumer change with no fallback behind it, and +it breaks the only test file that builds `CFDataVariable`s out of bare mocks. +`lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py` +does `CFDataVariable("var1", mocker.MagicMock(standard_name="x_wind"))`, +which puts `standard_name` on the *object*. `MagicMock.ncattrs()` returns a +`MagicMock`, and `list()` of one is empty, so after Task 6 the attribute +mapping for every double here is `{}`. Until now that did not show, because +`__getattr__` fell through to the netCDF4 member and found the mock's +attribute; Step 5 stops asking `__getattr__` at all. + +Give them the interface a netCDF variable actually has. Add above `class +Test`: + +```python +def _data_variable(mocker, name, **attributes): + """Build a CFDataVariable whose CF attributes read as a file's would. + + `MagicMock(standard_name=...)` puts the name on the object, which is no + longer where CFVariable looks for it. Drive `ncattrs`/`getncattr` + instead. + + """ + nc_var = mocker.MagicMock() + nc_var.ncattrs.return_value = list(attributes) + nc_var.getncattr.side_effect = attributes.__getitem__ + return CFDataVariable(name, nc_var) +``` + +and use it in `_setup`: + +```python + @pytest.fixture(autouse=True) + def _setup(self, mocker): + self.data_variables = [ + _data_variable(mocker, "var1", standard_name="x_wind"), + _data_variable(mocker, "var2", standard_name="y_wind"), + _data_variable(mocker, "var1", long_name="x component of wind"), + _data_variable( + mocker, + "var1", + standard_name="x_wind", + long_name="x component of wind", + ), + _data_variable(mocker, "var1"), + ] +``` + +and in `test_multiple_constraints__multiname`: + +```python + vars = self.data_variables + [ + _data_variable(mocker, "var1", standard_name="x_wind"), + _data_variable(mocker, "var1", standard_name="air_pressure"), + ] +``` + +Every assertion in the file is untouched — the expected result lists stay +exactly as they are. That is the point: the doubles change, the behaviour +they pin does not. Task 11 Step 10 revisits this helper once `cf_data` stops +being a netCDF4 object at all. + +- [ ] **Step 8: Run the loader tests** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/` +Expected: PASS, 0 failed, 1 skipped (§6.2). + +If `test__translate_constraints_to_var_callback.py` is red here with every +expected list reading `[False, False, ...]`, Step 7's helper was not applied +to one of the two construction sites. + +- [ ] **Step 9: Commit the loader** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/loader.py \ + lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py +git add lib/iris/fileformats/netcdf/loader.py \ + lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py +git commit -m "$(cat <<'EOF' +Read CF attributes from cf_var.attributes in the netCDF loader + +Seven sites. The constraint fast path also loses a computed attribute +name and a comment about a CFVariable property cache that no longer +exists. + +Its test doubles were MagicMocks carrying standard_name as a Python +attribute, which is not where CFVariable looks any more. They now drive +ncattrs/getncattr, like the netCDF variable they stand in for. No +assertion changed. + +The netCDF storage reaches in _get_cf_var_data - _data_array, datatype, +chunking(), the proxy class - are deliberately untouched: they need +cf_data to be a CFDatasetVariable, which is a later commit in this PR. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 10: `ugrid_load.py` — the plain reads** + +Lines 230, 370, 371, 383, 385 and 456 take Task 8's rule: + +```python + attr_climatology = coord_var.attributes.get("climatology") +``` +```python + edge_dimension = mesh_var.attributes.get("edge_dimension") + face_dimension = mesh_var.attributes.get("face_dimension") +``` +```python + elif coord.var_name in mesh_var.attributes.get("edge_coordinates", "").split(): +``` +```python + elif coord.var_name in mesh_var.attributes.get("face_coordinates", "").split(): +``` +```python + location = cf_var.attributes.get("location", "") +``` + +- [ ] **Step 11: `ugrid_load.py:322-326` — the `cf_role` probe and read** + +```python + cf_role_message = None + if "cf_role" not in mesh_var.attributes: + cf_role_message = f"{mesh_var.cf_name} has no cf_role attribute." + cf_role = "mesh_topology" + else: + cf_role = mesh_var.attributes["cf_role"] +``` + +- [ ] **Step 12: `ugrid_load.py:336-354` — the topology probes** + +```python + if "volume_node_connectivity" in mesh_var.attributes: + topology_dimension = 3 + elif "face_node_connectivity" in mesh_var.attributes: + topology_dimension = 2 + elif "edge_node_connectivity" in mesh_var.attributes: + topology_dimension = 1 + else: + # Nodes only. We aren't sure yet whether this is a valid option. + topology_dimension = 0 + + if "topology_dimension" not in mesh_var.attributes: + msg = ( + f"MeshXY variable {mesh_var.cf_name} has no 'topology_dimension'" + f" : *Assuming* topology_dimension={topology_dimension}" + ", consistent with the attached connectivities." + ) + warnings.warn(msg, category=_WarnComboCfDefaulting) + else: + quoted_topology_dimension = mesh_var.attributes["topology_dimension"] +``` + +These six probes are the reason `__contains__` records a read. Each names an +attribute Iris then acts on, and an unrecorded probe would put +`face_node_connectivity` onto every mesh cube as a stray attribute. + +- [ ] **Step 13: `ugrid_load.py:406` — the third computed-name site** + +`connectivity.cf_role` is a value, not a literal: + +```python + assert connectivity.var_name == mesh_var.attributes[connectivity.cf_role] +``` + +- [ ] **Step 14: Check no site was missed** + +Run: +```bash +grep -n "getattr(cf_var\|hasattr(cf_var\|getattr(mesh_var\|hasattr(mesh_var\|getattr(coord_var\|hasattr(coord_var" \ + lib/iris/fileformats/netcdf/ugrid_load.py +``` +Expected: no output. + +- [ ] **Step 15: Run the mesh tests** + +Run: +```bash +pytest lib/iris/tests/unit/fileformats/netcdf/test_ugrid_load.py \ + lib/iris/tests/unit/mesh/ lib/iris/tests/integration/mesh/ +``` +Expected: PASS, 0 failed, 0 skipped. + +Then the wider net, because this task's risk is stray cube attributes rather +than errors: + +Run: `pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/` +Expected: `0 failed, 26 skipped`, plus F14's ten `test_coord_systems.py` +errors and nothing else (§6.2). + +- [ ] **Step 16: Commit the mesh loader** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/ugrid_load.py +git add lib/iris/fileformats/netcdf/ugrid_load.py +git commit -m "$(cat <<'EOF' +Read CF attributes from cf_var.attributes in the UGRID loader + +Fourteen sites, including the third and last computed attribute name in +the loading path: getattr(mesh_var, connectivity.cf_role). + +The six hasattr probes around topology_dimension become "in +mesh_var.attributes", which still records the attribute as read - +otherwise face_node_connectivity and its neighbours would start +appearing as stray attributes on every mesh cube. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 10: `cf/_reader.py` reads — and writes — `.attributes` + +Eight sites, and the only module in the programme that *writes* CF attributes +during a load. The formula-terms code synthesises a `bounds` link onto a +variable that did not carry one: + +```python +cf_var.bounds = term_bounds_var.cf_name # _reader.py:324 +new_var.bounds = cf_var.bounds # _reader.py:341 +root_var.bounds = None # _reader.py:358 +``` + +Today those land in the instance `__dict__`, where they shadow anything +`__getattr__` would find — which is exactly how the matching +`hasattr(cf_variable, "bounds")` reads see them. Move the *reads* to +`.attributes` and leave the *writes* on the instance and the link becomes +invisible. So both move, together, in one commit. + +**Two `None` traps.** `cf_root_coord` at line 304 and `root_bounds_var` at +line 354 can each be `None`, and `getattr(None, "bounds", None)` answers +`None` quite happily where `None.attributes` raises. The comment at line +305 already warns about the first. A mechanical rewrite turns both into +`AttributeError` during load. + +**Files:** +- Modify: `lib/iris/fileformats/cf/_reader.py:304,309,324,339,341,350-358,439,460,497,499` +- Test: `lib/iris/tests/unit/fileformats/cf/test_CFReader.py` — unchanged +- Test: `lib/iris/tests/integration/test_cf.py` — unchanged + +**Interfaces:** +- Consumes: `CFVariable.attributes` from Task 6. +- Produces: nothing new. Leaves `_reader.py` free of `getattr`/`hasattr` over + `CFVariable`s, which Task 11 needs before the fallback can warn. + +**Deliberately left for Task 11**, because they read raw netCDF variables +rather than `CFVariable`s: `:204` `hasattr(variable, "mesh")`, `:235` +`getattr(nc_var, "grid_mapping", None)`, `:274` the global-attribute +comprehension and `_getncattr` itself. `:373` +`getattr(self.cf_group, "ugrid_coords", None)` stays as it is — `cf_group` is +a `CFGroup`, not a variable. + +- [ ] **Step 1: Write the failing test for the synthesised bounds link** + +This is the behaviour the task could silently lose, and no existing test +names it directly. In `lib/iris/tests/unit/fileformats/cf/test_CFReader.py`, +add: + +```python +class TestSynthesisedBoundsLink: + """The reader writes a bounds link that the file did not carry. + + _reader.py:324 attaches a formula term's bounds variable to the term. + Whatever stores that link has to be the same place the reads look, or + the link is written and never seen. + """ + + def test_a_written_bounds_link_is_visible_to_a_reader(self, mocker): + nc_var = mocker.MagicMock() + nc_var.ncattrs.return_value = ["units"] + nc_var.getncattr.side_effect = {"units": "m"}.__getitem__ + nc_var.dimensions = ("model_level_number",) + cf_var = CFAuxiliaryCoordinateVariable("a", nc_var) + + assert "bounds" not in cf_var.attributes + cf_var.attributes["bounds"] = "a_bnds" + + assert cf_var.attributes["bounds"] == "a_bnds" + assert "bounds" in cf_var.attributes + assert cf_var.bounds == "a_bnds" + + def test_a_written_bounds_link_overrides_the_file(self, mocker): + nc_var = mocker.MagicMock() + nc_var.ncattrs.return_value = ["bounds"] + nc_var.getncattr.side_effect = {"bounds": "from_file"}.__getitem__ + nc_var.dimensions = ("model_level_number",) + cf_var = CFAuxiliaryCoordinateVariable("a", nc_var) + + cf_var.attributes["bounds"] = "synthesised" + assert cf_var.bounds == "synthesised" + + def test_a_bounds_link_set_to_none_still_reads_as_present(self, mocker): + # _reader.py:358 invalidates a broken link by setting it to None + # rather than deleting it, and the reads downstream check presence + # before value. + nc_var = mocker.MagicMock() + nc_var.ncattrs.return_value = ["bounds"] + nc_var.getncattr.side_effect = {"bounds": "broken"}.__getitem__ + nc_var.dimensions = ("model_level_number",) + cf_var = CFAuxiliaryCoordinateVariable("a", nc_var) + + cf_var.attributes["bounds"] = None + assert "bounds" in cf_var.attributes + assert cf_var.attributes.get("bounds") is None +``` + +Add `CFAuxiliaryCoordinateVariable` to the module's imports from +`iris.fileformats.cf` if it is not already there. + +- [ ] **Step 2: Run it** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/test_CFReader.py -k Synthesised -v` +Expected: PASS on Task 6's implementation — `TrackedAttributes` is a +`MutableMapping` and `__getattr__` reads it. This is a characterisation test, +written now because the *next* step is what could break it. + +- [ ] **Step 3: Move the three writes onto the mapping** + +```python + if term_bounds_var != cf_var: + cf_var.attributes["bounds"] = term_bounds_var.cf_name +``` + +```python + if iris.FUTURE.derived_bounds and "bounds" in cf_var.attributes: + # Copy "old-style" derived bounds link + new_var.attributes["bounds"] = cf_var.attributes["bounds"] +``` + +and, in the `derived_bounds` invalidation block: + +```python + root_var.attributes["bounds"] = None +``` + +- [ ] **Step 4: Rewrite line 304, guarding the `None`** + +```python + # N.B. cf_root_coord may here be None, if the root var was not a + # coord - that is ok, it will not have a 'bounds', we will skip it. + root_bounds_name = None + if cf_root_coord is not None: + root_bounds_name = cf_root_coord.attributes.get("bounds") + if root_bounds_name in self.cf_group: + root_bounds_var = self.cf_group.get(root_bounds_name) + if "formula_terms" not in root_bounds_var.attributes: +``` + +The comment moves above the guard, where it now explains code rather than +trailing it. + +- [ ] **Step 5: Rewrite the `derived_bounds` invalidation block, guarding the second `None`** + +`self.cf_group.get(...)` returns `None` for a name the group does not hold, +and `getattr(None, "formula_terms", None)` answered `None` — which made the +`not` true and invalidated the link. Spell that out: + +```python + if iris.FUTURE.derived_bounds: + for cf_root in all_roots: + # Invalidate "broken" bounds connections + root_var = self.cf_group[cf_root] + if root_var.attributes.get("formula_terms") and root_var.attributes.get( + "bounds" + ): + root_bounds_var = self.cf_group.get(root_var.attributes["bounds"]) + if root_bounds_var is None or not root_bounds_var.attributes.get( + "formula_terms" + ): + # This means it is *not* a valid bounds var, according to CF, and so therefore we are + # invalidating the bounds. + root_var.attributes["bounds"] = None +``` + +- [ ] **Step 6: Rewrite the three remaining reads** + +Line 439, in `_build`: + +```python + if "bounds" in cf_variable.attributes: + bounds_name = cf_variable.attributes["bounds"] + if bounds_name not in cf_group: + bounds_var = self.cf_group.get(bounds_name) + if bounds_var: + # TODO: warning if span fails + if bounds_var.spans(cf_variable): + cf_group[bounds_name] = bounds_var +``` + +Line 460: + +```python + coordinates_attr = cf_variable.attributes.get("coordinates", "") +``` + +Lines 497-499: + +```python + if iris.FUTURE.derived_bounds: + if "standard_name" not in cf_root_var.attributes: + continue + name = cf_root_var.attributes["standard_name"] or cf_root_var.attributes.get( + "long_name" + ) +``` + +Note the asymmetry, which is deliberate and matches today: `standard_name` is +indexed, so a formula root without one raises — as it does today, with +`AttributeError` rather than `KeyError` — while `long_name` is fetched with +`get`, because it is only consulted when `standard_name` is falsy and a +variable may legitimately lack it. + +- [ ] **Step 7: Check no site was missed** + +Run: `grep -n "getattr(cf_\|hasattr(cf_\|\.bounds = " lib/iris/fileformats/cf/_reader.py` +Expected: only `getattr(self.cf_group, "ugrid_coords", None)` on line 373. + +- [ ] **Step 8: Run the reader tests** + +Run: +```bash +pytest lib/iris/tests/unit/fileformats/cf/ lib/iris/tests/integration/test_cf.py +``` +Expected: PASS, 0 failed, 1 skipped (§6.2). + +- [ ] **Step 9: Run the formula-terms and derived-bounds coverage** + +`iris.FUTURE.derived_bounds` gates three of the rewritten sites, so the +default-off suite does not reach them. + +Run: +```bash +pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/ \ + -k "derived or formula or hybrid or aux_factory" +``` +Expected: 0 failed. Then the full area: + +Run: `pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/` +Expected: `0 failed, 26 skipped`, plus F14's ten `test_coord_systems.py` +errors and nothing else (§6.2). + +- [ ] **Step 10: Commit** + +```bash +pre-commit run --files lib/iris/fileformats/cf/_reader.py \ + lib/iris/tests/unit/fileformats/cf/test_CFReader.py +git add lib/iris/fileformats/cf/_reader.py \ + lib/iris/tests/unit/fileformats/cf/test_CFReader.py +git commit -m "$(cat <<'EOF' +Read and write CF attributes through cf_var.attributes in CFReader + +CFReader is the one module that writes CF attributes during a load: the +formula-terms code synthesises a bounds link onto a variable that did not +carry one. Those writes land in the mapping now, alongside the reads +that look for them - moving one without the other would write a link +nothing could see. + +Two sites needed a None guard rather than a rewrite. cf_root_coord and +root_bounds_var can each be None, and getattr(None, "bounds", None) +answers None where None.attributes raises. The first already had a +comment saying so. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 11: The swap — `CFReader` builds a `NetCDFDataset` + +This is the task the previous five were clearing the way for. Every consumer +now reads CF attributes through `cf_var.attributes`; nothing reads them off a +netCDF4 object any more except `CFVariable.__init__` itself, the twelve +`identify()` calls, and `CFReader`'s own bookkeeping. Those move here, in one +step, and `cf_var.cf_data` stops being a `netCDF4.Variable` and becomes a +`CFDatasetVariable`. + +**One long red stretch.** A type swap cannot be done a consumer at a time: +the moment `CFReader` hands out `NetCDFDatasetVariable`s, everything that +still expects a netCDF4 object is broken, and the fix for each is in a +different file. Steps 3 to 10 therefore run with no test run between them, +and the green check is Step 11. This is deliberate, and it is why the five +preceding tasks exist — they shrink this stretch from "every consumer in the +package" to "the eight sites below". + +**Files:** +- Modify: `lib/iris/fileformats/cf/_variables.py:97-135` (`__init__`), + `:3319-3329` region (delete `_attributes_source`), `_deprecated_netcdf_member`, + and the twelve `identify()` attribute reads at + `:304, :353, :402, :478, :614, :687, :771, :874, :938, :1013, :1087, :1091` +- Modify: `lib/iris/fileformats/cf/_reader.py:132-186` (`__init__`), + `:200-206` (`_has_meshes`), `:235`, `:272-276`, `:543-549` (delete `_getncattr`) +- Modify: `lib/iris/fileformats/netcdf/_dataset.py` — `from_existing` gains + `warn_legacy_format`; `NetCDFDatasetVariable.deprecated_netcdf_member` starts + warning +- Modify: `lib/iris/fileformats/netcdf/loader.py:242, :254, :316, :321, :327-330` +- Test: `lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py` (create) +- Test: `lib/iris/tests/unit/fileformats/cf/identify_mixins.py`, + `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py`, + `lib/iris/tests/unit/fileformats/cf/test_CFReader.py`, + `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py`, + `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py` +- Test: the doubles that stand in for a netCDF4 variable, all in Step 10 — + `lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py`, + `lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py`, + `lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py` + +**Interfaces:** +- Consumes: `NetCDFDataset(location, mode="r", *, netcdf_format=None, + warn_legacy_format=False)`, `NetCDFDataset.from_existing(dataset)`, + `.location`, `.variables`, `.attributes`, `.close()` (Task 3); + `NetCDFDatasetVariable.name`, `.location`, `.dimensions`, `.shape`, + `.dtype`, `.size`, `.attributes`, `.chunking`, `.variable`, + `.is_variable_length`, `.is_emulated`, `.emulated_data_array`, + `.deprecated_netcdf_member(name)` (Task 2); `CFVariable.attributes` + (Task 6); every consumer rewritten in Tasks 8, 9 and 10. +- Produces: `CFVariable.cf_data` is a `CFDatasetVariable`, for every + `CFVariable` any part of Iris builds. `CFReader._dataset` is a + `NetCDFDataset`. `CFVariable.filename` comes from + `cf_data.location`. Task 12 relies on all of it, and on + `NetCDFDatasetVariable.deprecated_netcdf_member` warning. + +- [ ] **Step 1: Write the failing tests for the swap** + +Create `lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Tests for :class:`iris.fileformats.cf.CFReader` reading a real file. + +The rest of the CFReader tests drive it with mocks, which is the right shape +for its classification logic and the wrong shape for the question here: what +type the reader actually hands out, and what a load does with it. These use a +small file on disk instead. +""" + +import numpy as np +import pytest + +import iris +from iris.fileformats.cf import CFReader +from iris.fileformats.cf.dataset import CFDatasetVariable +from iris.fileformats.netcdf import _dataset, _thread_safe_nc + +UNREAD_COMMENT = "an attribute nothing in Iris reads" + + +@pytest.fixture +def sample_path(tmp_path): + """Write a small CF file: a data variable, a coordinate and its bounds.""" + path = tmp_path / "sample.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w") + dataset.title = "a sample file" + dataset.createDimension("time", 3) + dataset.createDimension("bnds", 2) + + time = dataset.createVariable("time", "f8", ("time",)) + time.standard_name = "time" + time.units = "days since 1970-01-01" + time.bounds = "time_bnds" + time.comment = UNREAD_COMMENT + time[:] = np.arange(3, dtype="f8") + + bounds = dataset.createVariable("time_bnds", "f8", ("time", "bnds")) + bounds[:] = np.zeros((3, 2)) + + air = dataset.createVariable("air", "f4", ("time",)) + air.standard_name = "air_temperature" + air.units = "K" + air.coordinates = "time" + air.comment = UNREAD_COMMENT + air[:] = np.arange(3, dtype="f4") + + dataset.close() + return path + + +class TestTheSwap: + def test_the_reader_owns_a_netcdf_dataset(self, sample_path): + with CFReader(str(sample_path)) as reader: + assert isinstance(reader._dataset, _dataset.NetCDFDataset) + + def test_cf_data_is_a_cf_dataset_variable(self, sample_path): + # The whole point of the PR: nothing downstream of here needs to know + # the file is netCDF. + with CFReader(str(sample_path)) as reader: + assert isinstance(reader.cf_group["air"].cf_data, CFDatasetVariable) + + def test_attributes_come_from_the_file(self, sample_path): + with CFReader(str(sample_path)) as reader: + air = reader.cf_group["air"] + assert air.attributes["units"] == "K" + assert air.units == "K" + assert air.dimensions == ("time",) + + def test_global_attributes_come_from_the_dataset(self, sample_path): + with CFReader(str(sample_path)) as reader: + assert reader.cf_group.global_attributes["title"] == "a sample file" + + def test_filename_is_the_variables_location(self, sample_path): + with CFReader(str(sample_path)) as reader: + assert reader.cf_group["air"].filename == str(sample_path) + + def test_the_bounds_variable_was_classified(self, sample_path): + # identify() now reads through .attributes, so a miss here means the + # classification pass lost sight of the file's attributes entirely. + with CFReader(str(sample_path)) as reader: + assert "time_bnds" in reader.cf_group.bounds + + def test_a_borrowed_dataset_is_used_and_not_closed(self, sample_path): + raw = _thread_safe_nc.DatasetWrapper(sample_path, mode="r") + try: + with CFReader(raw) as reader: + assert reader.cf_group["air"].units == "K" + assert raw.isopen() + finally: + raw.close() + + +class TestAttributesAreASnapshot: + """Finding F11. + + ``NetCDFDatasetVariable.attributes`` writes through to the file, which is + right for the saver and wrong for the reader: CFReader synthesises a + "bounds" link during load, on a file it opened read-only. CFVariable + therefore takes a copy. + """ + + def test_a_write_does_not_reach_the_file(self, sample_path): + with CFReader(str(sample_path)) as reader: + air = reader.cf_group["air"] + air.attributes["bounds"] = "invented" + + assert "bounds" not in air.cf_data.attributes + + +class TestAttributeTrackingAcrossReset: + """Review Focus 3. + + CFReader reads attributes while classifying variables, then calls + ``cf_attrs_reset()`` so the rules start from a clean record. Until this + PR, ``__getattr__`` cached the value on the instance, so a read after the + reset found the cache and was never recorded - and an attribute Iris had + in fact consumed was still reported unused, and so was copied onto the + cube as if the file had volunteered it. + """ + + def test_the_reader_resets_what_it_read_while_classifying(self, sample_path): + with CFReader(str(sample_path)) as reader: + time = reader.cf_group["time"] + # CFBoundaryVariable.identify() read "bounds" during __init__. + # Ask through .untracked, so that asking does not itself record. + assert "bounds" in time.attributes.untracked + assert dict(time.cf_attrs_used()) == dict(time.cf_attrs_ignored()) + + def test_a_read_after_the_reset_is_recorded(self, sample_path): + with CFReader(str(sample_path)) as reader: + time = reader.cf_group["time"] + + assert time.units == "days since 1970-01-01" + assert "units" in dict(time.cf_attrs_used()) + assert "units" not in dict(time.cf_attrs_unused()) + + def test_only_unread_attributes_reach_the_cube(self, sample_path): + cube = iris.load_cube(str(sample_path)) + + assert cube.attributes == {"comment": UNREAD_COMMENT} + assert cube.coord("time").attributes == {"comment": UNREAD_COMMENT} +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py` +Expected: `TestTheSwap::test_the_reader_owns_a_netcdf_dataset` and +`test_cf_data_is_a_cf_dataset_variable` FAIL on `assert isinstance(...)` — +the reader still opens an `EncodedDataset` directly. `TestAttributesAreASnapshot` +fails with `AttributeError: 'Variable' object has no attribute 'attributes'`. +The rest pass already: they describe behaviour this task must preserve. + +- [ ] **Step 3: Take `CFVariable.__init__` off the netCDF4 API** + +In `lib/iris/fileformats/cf/_variables.py`, delete `_attributes_source` +entirely — the function, its docstring and its comment. Then in +`CFVariable.__init__`, replace the two lines it fed: + +```python + self.attributes = TrackedAttributes( + dict(data.attributes), ignored=_CF_ATTRS_IGNORE + ) +``` + +`dict(...)`, not the mapping itself. `NetCDFDatasetVariable.attributes` is a +live view: `__setitem__` calls `setncattr` on the file. CFReader writes a +synthesised `bounds` link through `cf_var.attributes` (Task 10, Step 3) on a +file opened `mode="r"`, and a live view would turn that into +`RuntimeError: NetCDF: Write to read only`. Copying also makes the mapping +cheap to re-read, which is what `cf_attrs_unused()` asks of it once per +variable per load. See finding F11. + +Replace the `filename` block immediately below it: + +```python + """File source of the NetCDF content.""" + try: + self.filename = data.location + except AttributeError: + self.filename = "" +``` + +`CFDatasetVariable.location` is the path or URL of the dataset the variable +belongs to — the same string `data.group().filepath()` used to return, and +now supplied by the object itself rather than reached around it. The +`try/except` stays because the identify() test stubs are not +`CFDatasetVariable`s and have no `location`; they keep the same +`""` they get today. + +Finally, point the compatibility route at the storage object. Replace the +body of `_deprecated_netcdf_member`, and its transitional comment: + +```python + def _deprecated_netcdf_member(self, name): + """Return a netCDF4 member of the backing variable, as this class used to. + + The one-cycle compatibility route for code that reached netCDF4 API + through a CFVariable. Records nothing: this is not a CF attribute. + The storage object decides what this means, and warns - see + :meth:`iris.fileformats.netcdf._dataset.NetCDFDatasetVariable.deprecated_netcdf_member`. + + """ + return self.cf_data.deprecated_netcdf_member(name) +``` + +- [ ] **Step 4: Move the twelve `identify()` reads onto `.attributes`** + +Still in `_variables.py`. `identify()` is handed the dataset's variables, +which are now `CFDatasetVariable`s, so `getattr(nc_var, ...)` no longer +reaches a CF attribute at all — it reaches the storage class's own members, +or nothing. + +At `:304, :353, :402, :478, :614, :687, :771` and `:874`, each reading: + +```python + nc_var_att = getattr(nc_var, cls.cf_identity, None) +``` + +becomes: + +```python + nc_var_att = nc_var.attributes.get(cls.cf_identity) +``` + +At `:938` and `:1013`, inside the loop over `cls.cf_identities`: + +```python + nc_var_att = getattr(nc_var, identity, None) +``` + +becomes: + +```python + nc_var_att = nc_var.attributes.get(identity) +``` + +At `:1087`: + +```python + is_mesh_var = nc_var.attributes.get("cf_role", "") == "mesh_topology" +``` + +At `:1091`: + +```python + nc_var_att = nc_var.attributes.get(cls.cf_identity) +``` + +These reads are deliberately **not** tracked. `CFDatasetVariable.attributes` +is a plain mapping; tracking lives in `CFVariable.attributes`, and these run +before any `CFVariable` exists. That matches what they do today — `getattr` +on a raw netCDF variable recorded nothing — and it is why `CFReader._reset` +still has work to do afterwards. + +- [ ] **Step 5: Make `CFReader` build a dataset** + +In `lib/iris/fileformats/cf/_reader.py`, replace `__init__` down to the line +that sets `self._check_monotonic`: + +```python + def __init__(self, file_source, warn=False, monotonic=False): + # Ensure safe operation for destructor, should init fail. + self._own_file = False + if isinstance(file_source, str): + # Create from filepath : open it + own it (=close when we die). + if not urlparse(file_source).scheme: + self._filename = Path(file_source).expanduser() + else: + self._filename = file_source + + self._dataset = NetCDFDataset( + self._filename, mode="r", warn_legacy_format=warn + ) + self._own_file = True + else: + # We have been passed an open dataset. + # We use it but don't own it (don't close it). + self._dataset = NetCDFDataset.from_existing( + file_source, warn_legacy_format=warn + ) + self._filename = self._dataset.location + + #: Collection of CF-netCDF variables associated with this netCDF file + self.cf_group = self.CFGroup() + + # Result of parsing "grid_mapping" attribute; mapping of coordinate_system => coordinates + self._coord_system_mappings = {} + + self._check_monotonic = monotonic +``` + +Four things go, and all four went somewhere: the choice between +`EncodedDataset` and `DatasetWrapper` on `DECODE_TO_STRINGS_ON_READ`, the +`set_auto_chartostring(False)` call, the netCDF3 warning, and +`self._dataset.filepath()`. `NetCDFDataset` does all of them, which is what +§3's "single home for netCDF API" line means in practice. + +Leave the block that follows untouched apart from its two dead lines: + +```python + self._with_ugrid = True + if not self._has_meshes(): + self._trim_ugrid_variable_types() + self._with_ugrid = False + + # Read the variables in the dataset only once to reduce runtime. + # NetCDFDataset.variables caches, so this is the same dict - and the + # same CFDatasetVariable objects - that _has_meshes just walked. + variables = self._dataset.variables + self._translate(variables) + self._build_cf_groups(variables) + self._reset(variables) +``` + +`from_existing` needs the new keyword. In +`lib/iris/fileformats/netcdf/_dataset.py`, give both constructors the same +warning by factoring it out. Replace the warning block in `__init__` with a +call: + +```python + if warn_legacy_format: + self._warn_if_legacy_format() +``` + +and add the method, beside `close`: + +```python + def _warn_if_legacy_format(self) -> None: + """Warn that this file would load faster in netCDF4 format, if it would.""" + if self._dataset.file_format in _LEGACY_FORMATS: + warnings.warn( + "Optimise CF-netCDF loading by converting data from NetCDF3 " + 'to NetCDF4 file format using the "nccopy" command.', + category=iris.warnings.IrisLoadWarning, + ) +``` + +Then extend `from_existing`: + +```python + @classmethod + def from_existing(cls, dataset, *, warn_legacy_format: bool = False): +``` + +and end its body, before the `return`: + +```python + if warn_legacy_format: + instance._warn_if_legacy_format() +``` + +CFReader has always offered this warning for a borrowed dataset as well as +for one it opened, and the constraint on this PR is no behaviour change. +Add the matching test to `TestLegacyFormatWarning` in +`lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py`: + +```python + def test_warns_for_a_borrowed_dataset_when_asked(self, tmp_path): + legacy = _write_netcdf3(tmp_path) + raw = _thread_safe_nc.DatasetWrapper(legacy, mode="r") + try: + with pytest.warns(IrisLoadWarning, match="nccopy"): + _dataset.NetCDFDataset.from_existing(raw, warn_legacy_format=True) + finally: + raw.close() +``` + +- [ ] **Step 6: Finish `_reader.py` — meshes, grid mapping, globals, `_getncattr`** + +`_has_meshes` walks raw variables today. Replace it: + +```python + def _has_meshes(self): + result = False + for variable in self._dataset.variables.values(): + attributes = variable.attributes + if "mesh" in attributes or "node_coordinates" in attributes: + result = True + break + return result +``` + +At `:235`, in `_translate`: + +```python + if grid_mapping_attr := nc_var.attributes.get("grid_mapping"): +``` + +At `:272-276`, the global attributes. `_getncattr` existed to give +`getncattr` a `getattr`-shaped default, and `_NetCDFAttributes` already +absorbs the malformed-file case it was tolerating (Task 2): + +```python + # Identify global netCDF attributes. + self.cf_group.global_attributes.update(dict(self._dataset.attributes)) +``` + +Delete `_getncattr` at the foot of the module, and its docstring. The +`test___init__.py` parametrisation that names `_getncattr` stays as it is: +it asserts the name is *not* on the package, which a deleted name satisfies, +and it goes on guarding against the name coming back. + +Finally the imports. `_reader.py` no longer needs `_bytecoding_datasets` or +`_thread_safe_nc`; it needs the dataset instead: + +```python +from iris.fileformats.netcdf._dataset import NetCDFDataset +``` + +Check what is left: `warnings` is still used by the `grid_mapping` parse +error, `iris.warnings` with it. Run +`pre-commit run ruff --files lib/iris/fileformats/cf/_reader.py` at the end +of the task and let F401 name anything that has gone cold. + +- [ ] **Step 7: Move the loader's five storage reads onto the interface** + +These are the sites Task 9 deliberately left alone: they ask about *storage*, +not about CF attributes, and until now the only way to ask was to reach for +netCDF4 API. In `lib/iris/fileformats/netcdf/loader.py`: + +`:242`, the Xarray bridge (#4994) — + +```python + if cf_var.cf_data.is_emulated: + # The variable is not an actual netCDF4 file variable, but an emulating + # object with an attached data array (either numpy or dask), which can be + # returned immediately as-is. This is used as a hook to translate data to/from + # netcdf data container objects in other packages, such as xarray. + # See https://github.com/SciTools/iris/issues/4994 "Xarray bridge". + result = cf_var.cf_data.emulated_data_array +``` + +`:254`, the VLEN check — + +```python + if cf_var.cf_data.is_variable_length: +``` + +`_thread_safe_nc.VLType` leaves the loader with it; the whole `isinstance` +disappears. + +`:316` and `:321`, the data proxy. `cf_var.cf_data` is no longer the netCDF +variable the proxy needs, so both ask the interface for it: + +```python + if isinstance(cf_var.cf_data.variable, _bytecoding_datasets.EncodedVariable): + proxy_class = _bytecoding_datasets.EncodedNetCDFDataProxy + else: + proxy_class = _thread_safe_nc.NetCDFDataProxy + + proxy = proxy_class( + cf_var.cf_data.variable, dtype, cf_var.filename, fill_value + ) +``` + +`.variable` is the one deliberate escape hatch in the interface, and the +proxy is what it exists for: it is pickled and shipped to Dask workers, and +it has to carry a netCDF handle to reopen. Task 12 uses it for the same +reason. A Zarr backend will supply its own proxy class here in a later PR. + +`:327-330`, the chunking. `NetCDFDatasetVariable.chunking` is a property, and +it has already turned netCDF's two ways of saying "unchunked" — `None` for a +non-version-4 file, the string `"contiguous"` for a version-4 one — into one. +That collapses the two-step dance the loader used to do: + +```python + chunks = cf_var.cf_data.chunking + if chunks is None: + # Unchunked : either a non-version-4 file, or a contiguous + # version-4 variable. Neither offers a chunking to adopt. + if ( + CHUNK_CONTROL.mode is ChunkControl.Modes.FROM_FILE + and isinstance(cf_var, iris.fileformats.cf.CFDataVariable) + ): + raise KeyError( + f"{cf_var.cf_name} does not contain pre-existing chunk specifications." + f" Instead, you might wish to use CHUNK_CONTROL.set(), or just use default" + f" behaviour outside of a context manager. " + ) + # Equivalent to chunks=None, but value required by chunking control + chunks = list(cf_var.shape) + else: + # The chunk-control block below assigns into this. + chunks = list(chunks) +``` + +Delete the two lines that set and then test `chunks = "contiguous"`; the +`if chunks == "contiguous":` line and its comment go with them. `list(chunks)` +matters: the property returns a tuple, and +`chunks[i_dim] = dim_chunksize` a few lines further down needs a list. + +- [ ] **Step 8: Teach the identify() stubs the new shape** + +The stubs in the CF unit tests stand in for a variable that `identify()` +reads and `CFVariable.__init__` copies from. Both now go through +`attributes`. + +In `lib/iris/tests/unit/fileformats/cf/identify_mixins.py`, delete the +`getncattr` method Task 6 added to `_NetCDFVar`, and add in its place: + +```python + @property + def attributes(self): + """The CF attributes, as a CFDatasetVariable presents them.""" + return {name: getattr(self, name) for name in self.ncattrs()} +``` + +`ncattrs()` stays: it is what decides which of the instance's members count +as file attributes, and the property is built from it. Neither `attributes` +nor `ncattrs` appears in `self.__dict__`, so neither can list itself. + +In `lib/iris/tests/unit/fileformats/cf/test_CFCoordinateVariable.py`, add the +same to `_CoordVariableStub`, whose `ncattrs()` returns `[]`: + +```python + @property + def attributes(self): + return {} +``` + +That stub had needed nothing until now, because `getattr` on it answered +whatever the test had set. `dict(data.attributes)` will not invent a mapping. + +- [ ] **Step 9: Rewrite the `test_CFVariable.py` doubles** + +The module's `MagicMock` variables have to present the same surface. Replace +`make_nc_var` and the fixture below it: + +```python +def make_nc_var(mocker): + nc_var = mocker.MagicMock() + nc_var.attributes = { + "coordinates": "x y", + "standard_name": "air_temperature", + "_FillValue": -999, + } + nc_var.dimensions = ("time", "lat") + nc_var.location = "/tmp/file.nc" + nc_var.__len__.return_value = 4 + nc_var.__getitem__.return_value = "payload" + + return nc_var + + +@pytest.fixture +def nc_var(mocker): + return make_nc_var(mocker) + + +@pytest.fixture +def nc_var_without_location(nc_var): + del nc_var.location + + return nc_var +``` + +`TestInit` follows the rename: + +```python +class TestInit: + def test_records_filename_from_the_variables_location(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.filename == "/tmp/file.nc" + assert cf_var.cf_name == "foo" + assert cf_var.cf_data is nc_var + assert cf_var.cf_group is None + assert cf_var.cf_terms_by_root == {} + assert cf_var._to_be_promoted is False + + def test_falls_back_to_unknown_filename_without_a_location( + self, nc_var_without_location + ): + cf_var = CFVariableSub("foo", nc_var_without_location) + + assert cf_var.filename == "" +``` + +Then five tests that named `ncattrs`/`getncattr`. In `TestAttributeAccess`, +replace `test_attributes_are_read_from_the_file_once` — there is no call to +count any more, and the property it was really asserting is better said +directly: + +```python + def test_attributes_are_a_snapshot_taken_at_construction(self, nc_var): + # Finding F11. Copied, not wrapped: the storage object's mapping + # writes through to the file, and CFReader synthesises a "bounds" + # link on a file it opened read-only. + cf_var = CFVariableSub("foo", nc_var) + + nc_var.attributes["units"] = "K" + assert "units" not in cf_var.attributes.untracked + + cf_var.attributes["bounds"] = "foo_bnds" + assert "bounds" not in nc_var.attributes +``` + +Replace `test_getattr_of_a_non_attribute_reaches_the_variable`: + +```python + def test_getattr_of_a_non_attribute_reaches_the_variable(self, nc_var): + # The one-cycle compatibility route, now delegated to the storage + # object - which is also what warns. See Step 15. + nc_var.deprecated_netcdf_member.return_value = 42 + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.not_an_ncattr == 42 + nc_var.deprecated_netcdf_member.assert_called_once_with("not_an_ncattr") + assert "not_an_ncattr" not in dict(cf_var.cf_attrs()) + assert "not_an_ncattr" not in dict(cf_var.cf_attrs_used()) +``` + +and `test_getattr_of_nothing_at_all_raises_attribute_error`, which used to +swap in a bare `object()` to get a real miss: + +```python + def test_getattr_of_nothing_at_all_raises_attribute_error(self, nc_var): + nc_var.deprecated_netcdf_member.side_effect = AttributeError("nonesuch") + cf_var = CFVariableSub("foo", nc_var) + + with pytest.raises(AttributeError, match="nonesuch"): + cf_var.nonesuch +``` + +In `TestTypedProperties`, the shadowing check: + +```python + def test_a_file_attribute_does_not_reach_the_typed_properties(self, nc_var): + # The properties are looked up before __getattr__ runs, so a file + # attribute called "shape" cannot displace the variable's shape. + nc_var.attributes = {"shape": "not a shape"} + nc_var.shape = (2,) + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.shape == (2,) + assert cf_var.attributes["shape"] == "not a shape" +``` + +In `TestShadowedAttributeNames`, the fixture: + +```python + @pytest.fixture + def shadowing(self, nc_var): + nc_var.attributes = { + name: f"file value of {name}" for name in SHADOWED_NAMES + } | {"units": "K"} + return CFVariableSub("foo", nc_var) +``` + +In `TestGetattrAndHasattr`, the dunder probe: + +```python + def test_dunder_probes_do_not_reach_the_file(self, nc_var): + # copy, pickle and numpy all probe for dunders. Answering one from + # file data would make a CFVariable behave as whatever the file says. + nc_var.attributes = {"__array__": "nonsense"} + cf_var = CFVariableSub("foo", nc_var) + + assert not hasattr(cf_var, "__array_interface__") + assert cf_var.attributes.untracked["__array__"] == "nonsense" +``` + +`test_hasattr_false_marks_nothing`, which sets `cf_var.cf_data = object()`, +needs nothing: `object()` has no `deprecated_netcdf_member` either, so the +miss is still a miss. + +- [ ] **Step 10: Rewrite the `test_CFReader.py` and constraint-callback doubles** + +`netcdf_variable()` builds a mock whose CF attributes are ordinary members, +because that is what `getattr`-based `identify()` read. The dataset now wraps +each of these in a `NetCDFDatasetVariable`, which reads `ncattrs()` and +`getncattr()`, so the two have to agree. In +`lib/iris/tests/unit/fileformats/cf/test_CFReader.py`, replace the +construction at the end of `netcdf_variable`: + +```python + members = dict( + ancillary_variables=ancillary_variables, + coordinates=coordinates, + bounds=bounds, + climatology=climatology, + formula_terms=formula_terms, + grid_mapping=grid_mapping, + cell_measures=cell_measures, + standard_name=standard_name, + **{name: None for name in ugrid_identities}, + ) + # A None member stood for "no such attribute" when identify() read these + # with getattr(..., None). Now that it reads a mapping, absent means + # absent - so a None must not be listed. + attributes = {key: value for key, value in members.items() if value is not None} + ncvar = mocker.Mock( + name=name, + dimensions=dimensions, + ndim=ndim, + dtype=dtype, + ncattrs=mocker.Mock(return_value=list(attributes)), + getncattr=mocker.Mock(side_effect=attributes.__getitem__), + **members, + ) + return ncvar +``` + +`Test_init_and_lifecycle` patches the class `CFReader` used to name, which it +no longer does. Two edits in `_setup` and one test: + +```python + self.encoded_ds = mocker.patch( + "iris.fileformats.netcdf._bytecoding_datasets.EncodedDataset", + return_value=self.dataset, + ) +``` + +stays exactly as it is — `NetCDFDataset` reaches the class by the same path, +so the patch still bites. But `test_init_uses_dataset_wrapper_when_string_decode_disabled` +names `_reader`'s own import, which has gone: + +```python + wrapper_ds = mocker.patch( + "iris.fileformats.netcdf._thread_safe_nc.DatasetWrapper", + return_value=self.dataset, + ) +``` + +and `self.dataset` needs the two members `NetCDFDataset` now asks of it, +beside the three it already has: + +```python + self.dataset = mocker.Mock( + file_format="NetCDF4", + variables=self.variables, + ncattrs=mocker.Mock(return_value=[]), + filepath=mocker.Mock(return_value="in-memory.nc"), + set_auto_chartostring=mocker.Mock(), + isopen=mocker.Mock(return_value=True), + ) +``` + +`test_init_with_no_meshes_trims_ugrid_variable_types` substitutes variables +that `_has_meshes` walks, and those are wrapped now. Give both of them the +netCDF surface a wrapper needs: + +```python + def test_init_with_no_meshes_trims_ugrid_variable_types(self, mocker): + self.dataset.variables = { + "a": netcdf_variable(mocker, "a", "x", np.float64), + "b": netcdf_variable(mocker, "b", "x", np.float64, mesh="my_mesh"), + } + + reader = CFReader("dummy.nc") + + assert reader._with_ugrid is True + + mesh_free = mocker.Mock( + file_format="NetCDF4", + variables={"a": netcdf_variable(mocker, "a", "x", np.float64)}, + ncattrs=mocker.Mock(return_value=[]), + filepath=mocker.Mock(return_value="in-memory.nc"), + set_auto_chartostring=mocker.Mock(), + ) + self.encoded_ds.return_value = mesh_free + reader = CFReader("dummy.nc") + + assert reader._with_ugrid is False + assert CFUGridMeshVariable not in reader._variable_types +``` + +`netcdf_variable` does not take a `mesh` keyword today; add one alongside +`grid_mapping`, defaulting to `None`, and include it in `members` above. +It is one of the UGRID identities, so it is already in the `ugrid_identities` +sweep — the explicit keyword just lets a test set it. + +`TestSynthesisedBoundsLink`, added in Task 10, builds `CFVariable`s straight +from `MagicMock`s. All three of its mocks change the same way — for example +the first: + +```python + nc_var = mocker.MagicMock() + nc_var.attributes = {"units": "m"} + nc_var.dimensions = ("model_level_number",) +``` + +with `{"bounds": "from_file"}` and `{"bounds": "broken"}` in the other two. + +`lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py` +next. Task 9 gave it a `_data_variable` helper when it rewrote the constraint +fast path; the helper's body becomes: + +```python +def _data_variable(mocker, name, **attributes): + """A CFDataVariable whose CF attributes are exactly the ones named.""" + return CFDataVariable(name, mocker.MagicMock(attributes=attributes)) +``` + +The call sites do not move — that is the whole reason Task 9 introduced a +helper instead of editing seven mock constructions twice. + +Then `lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py`, +which is the direct test of everything Step 7 rewrote. Its `_make` builds a +`cf_data` that answers netCDF4's API, and all four of those answers changed. +Replace the body (`:26-49`) with: + +```python + def _make(self, chunksizes=None, shape=None, dtype="i4", **extra_properties): + if shape is None: + shape = self.shape + cf_data = self.mocker.MagicMock( + spec=iris.fileformats.netcdf._dataset.NetCDFDatasetVariable, + fill_value=None, + dimensions=tuple("dim_" + str(x) for x in range(len(shape))), + shape=shape, + chunking=chunksizes, + is_variable_length=False, + is_emulated=False, + ) + if dtype is not str: # for testing VLen str arrays (dtype=`class `) + dtype = np.dtype(dtype) + cf_var = self.mocker.MagicMock( + spec=iris.fileformats.cf.CFVariable, + dtype=dtype, + cf_data=cf_data, + filename=self.filename, + cf_name="DUMMY_VAR", + shape=shape, + size=np.prod(shape), + **extra_properties, + ) + cf_var.__getitem__.return_value = self.mocker.sentinel.real_data_accessed + return cf_var +``` + +Four changes, each matching one line of Step 7: + +- `chunking` is a plain value, not `MagicMock(return_value=...)`. It is a + property now. The tests that pass `"contiguous"` — + `test_cf_data_contiguous` — must pass `None` instead, because the property + has already collapsed netCDF's two spellings of "unchunked" into one. Keep + the test; rename it `test_cf_data_unchunked` and merge it into + `test_cf_data_no_chunks`'s expectation. Its comment, "Chunks 'contiguous' + is equivalent to no chunks", becomes a comment on + `NetCDFDatasetVariable.chunking` in `_dataset.py`, where the collapsing now + happens — that is where a reader needs it. +- `is_variable_length` replaces `datatype`. The six `test_vltype__*` tests + pass `datatype=mock_vltype` as an `extra_properties` keyword, which lands on + `cf_var`, not `cf_data`. They become `_make(..., cf_data_is_variable_length=True)` + — or, less cleverly and better, set it directly: + +```python + def test_vltype__1000str_is_lazy(self): + cf_var = self._make(shape=(1000,), dtype=str) + cf_var.cf_data.is_variable_length = True + var_data = _get_cf_var_data(cf_var) + assert isinstance(var_data, da.Array) +``` + + and the `VLType` import goes with the last of them. +- `test_cf_data_emulation` passes `_data_array=emulated_data`, which likewise + landed on `cf_var`. It becomes: + +```python + def test_cf_data_emulation(self, mocker): + # Check that a variable emulation object passes its real data directly. + emulated_data = mocker.Mock() + cf_var = self._make(chunksizes=None) + cf_var.cf_data.is_emulated = True + cf_var.cf_data.emulated_data_array = emulated_data + result = _get_cf_var_data(cf_var) + # This should get directly returned. + assert emulated_data is result +``` + +- `spec=` on `cf_data` is the point of the whole edit. Without it, every one + of these reads returns a fresh `MagicMock` and every test passes whatever + Step 7 wrote. With it, a member Step 7 invented but did not implement is an + `AttributeError` here, in the test that exists to catch it. + +`_FillValue` becomes `fill_value`: the interface's name for it. It is never +read in this file, but leaving a netCDF4 spelling on a spec'd mock is a +`AttributeError` waiting for whoever adds the next test. + +Last, `lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py`. +Two mocks there carry `cf_data=Mock(spec=[])` with a following +`delattr(..., "_data_array")` (`:60` and `:98`, the `delattr`s at `:77` and +`:111`), which was how a test said "this is a real file variable, not an +emulated one". `spec=[]` means the mock has *no* attributes, so +`cf_var.cf_data.is_emulated` raises. Both become: + +```python + cf_data = self.mocker.Mock(spec=["is_emulated"], is_emulated=False) +``` + +and both `delattr(..., "_data_array")` lines go. The `del cf_data.flag_values` +trio above the first one stays and still works — `Mock.__delattr__` tolerates +deleting a name the spec never had **[verified]**. + +- [ ] **Step 11: Run the swap's own tests, then the two packages** + +Run: `pytest lib/iris/tests/unit/fileformats/cf/ lib/iris/tests/unit/fileformats/netcdf/` +Expected: PASS, 0 failed, **2 skipped** — the two pre-existing skips §6.2 +records, one in each directory. This is the first test run since Step 2, and +the first point at which the tree is consistent again. + +If `test_CFReader.py` fails in bulk with `KeyError` from `getncattr`, the +`netcdf_variable` rewrite and the mock's members have fallen out of step — +`ncattrs()` is listing a name `getncattr` cannot supply. If a single +`identify()` test fails, compare its stub against Step 8. + +- [ ] **Step 12: Run the load path end to end, then commit** + +Run: `pytest -n auto lib/iris/tests/unit/ lib/iris/tests/integration/` +Expected: matches the baseline — §6.3's diff is empty and the skip count has +not risen. Concretely: `lib/iris/tests/unit/` alone is `7540 passed, +35 skipped, 0 failed` at baseline **[verified]** — the 35 are gdal, +`iris_grib`, cartopy and FUTURE-context guards spread over seven modules, not +a smell — and the integration half is `0 failed` with only F14's ten +`test_coord_systems.py` errors. A new skip here means a test stopped running +rather than started passing; that is the failure mode this step exists to +catch, and it is the one the tail line will not shout about. + +```bash +pre-commit run --files lib/iris/fileformats/cf/_variables.py \ + lib/iris/fileformats/cf/_reader.py \ + lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/fileformats/netcdf/loader.py \ + lib/iris/tests/unit/fileformats/cf/identify_mixins.py \ + lib/iris/tests/unit/fileformats/cf/test_CFCoordinateVariable.py \ + lib/iris/tests/unit/fileformats/cf/test_CFVariable.py \ + lib/iris/tests/unit/fileformats/cf/test_CFReader.py \ + lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py \ + lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py \ + lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py \ + lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py +git add lib/iris/fileformats/cf/ lib/iris/fileformats/netcdf/ \ + lib/iris/tests/unit/fileformats/cf/ lib/iris/tests/unit/fileformats/netcdf/ \ + lib/iris/tests/unit/fileformats/nc_load_rules/ +git commit -m "$(cat <<'EOF' +Build a NetCDFDataset in CFReader, so cf_data is a CFDatasetVariable + +The swap this PR was for. CFReader opened a netCDF dataset and handed +its variables straight out; every CFVariable therefore wrapped a +netCDF4.Variable, and so did every consumer downstream. It now opens a +NetCDFDataset, and cf_data is a CFDatasetVariable. + +Nothing outside iris.fileformats.netcdf calls netCDF4 API any more. The +twelve identify() reads, CFVariable.__init__, CFReader's mesh probe, +grid-mapping parse and global attributes, and the loader's five storage +questions all go through the interface instead. _getncattr is gone: +_NetCDFAttributes absorbs the malformed-file case it existed for. + +CFVariable copies the attribute mapping rather than wrapping it. The +storage object's mapping writes through to the file, and CFReader +synthesises a bounds link during load, on a file opened read-only. + +The test doubles that stood in for a netCDF4 variable stand in for a +NetCDFDatasetVariable now, and the ones in test__get_cf_var_data.py gain +a spec= so that an invented member is an error rather than a mock. No +assertion changed. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 13: Write the failing test for the deprecation warning** + +Nothing in Iris reaches netCDF4 API through a `CFVariable` any more, so the +route can start telling third-party code that it is going away. It could not +have warned before this point: Tasks 6 to 10 left consumers on it one at a +time, and a warning would have fired thousands of times per load. + +In `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py`, +replace the two tests Task 2 wrote for this method: + +```python +class TestDeprecatedNetcdfMember: + def test_a_reach_through_warns(self, air): + with pytest.warns(IrisDeprecation, match="ncattrs"): + names = air.deprecated_netcdf_member("ncattrs")() + assert sorted(names) == ["long_name", "standard_name", "units"] + + def test_a_missing_name_raises_and_does_not_warn(self, air): + # hasattr() probes land here, and a probe that comes back False is + # not a use of anything - warning about it would be noise. + with warnings.catch_warnings(): + warnings.simplefilter("error") + with pytest.raises(AttributeError, match="no_such_netcdf_member"): + air.deprecated_netcdf_member("no_such_netcdf_member") + + def test_the_message_names_the_replacement(self, air): + with pytest.warns(IrisDeprecation, match=r"cf_data\.variable"): + air.deprecated_netcdf_member("ncattrs") +``` + +Add to the module's imports: + +```python +import warnings + +from iris._deprecation import IrisDeprecation +``` + +Adjust `sorted(names)` to whatever the module's `air` fixture actually +carries if it differs — Task 2 wrote that fixture, and this assertion is +lifted from its own `test_deprecated_netcdf_member_reaches_the_wrapper`. + +- [ ] **Step 14: Run them to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py -k Deprecated -v` +Expected: `test_a_reach_through_warns` and `test_the_message_names_the_replacement` +FAIL with `DID NOT WARN`. `test_a_missing_name_raises_and_does_not_warn` +passes already. + +- [ ] **Step 15: Make the reach-through warn** + +In `lib/iris/fileformats/netcdf/_dataset.py`, replace +`NetCDFDatasetVariable.deprecated_netcdf_member`: + +```python + def deprecated_netcdf_member(self, name: str) -> Any: + """Return a member of the backing netCDF variable wrapper, with a warning.""" + # Fetch before warning, so that a name the wrapper does not have is an + # ordinary AttributeError. hasattr() probes arrive here, and a probe + # that comes back False has not used anything. + value = getattr(self._variable, name) + warn_deprecated( + f"Reaching netCDF variable member {name!r} through a CFVariable is " + "deprecated and will be removed in a future release. Use the CF " + "dataset interface instead - iris.fileformats.cf.dataset - or, for " + "a netCDF-only need, cf_var.cf_data.variable." + ) + return value +``` + +Add to the module's imports: + +```python +from iris._deprecation import warn_deprecated +``` + +The base class's `CFDatasetVariable.deprecated_netcdf_member` (Task 1) keeps +raising `AttributeError`: a backend with no netCDF underneath it has nothing +to deprecate, and nothing to warn about. + +- [ ] **Step 16: Run them to verify they pass, and that nothing in Iris trips it** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py -k Deprecated -v` +Expected: PASS. + +Then prove the route is unused, which is the claim that entitles it to warn: + +Run: `pytest -W "error:Reaching netCDF variable member" lib/iris/tests/unit/fileformats/cf/ lib/iris/tests/unit/fileformats/netcdf/ lib/iris/tests/integration/` +Expected: PASS. Any failure here means a site in `lib/iris/` still reaches +through — find it in the traceback and move it onto `.attributes` or onto a +named member of the interface, in this task. + +(Corrected in Task 13. As first written this step said +`-W error::iris._deprecation.IrisDeprecation`, which cannot run at all: it +exits 4 during conftest import, because `lib/iris/tests/graphics/__init__.py` +calls `warn_deprecated` for `GraphicsTestMixin` at import time. That file and +`lib/iris/_deprecation.py` are byte-identical to the merge base, so it is +pre-existing and unrelated to this PR — and it means the check as written was +never runnable. Task 13 Step 7 carries the same correction; this copy was +missed when that one was made. The message-targeted filter is narrower *and* +stronger: it turns only this PR's own reach-through warning into an error, so +an unrelated pre-existing deprecation can neither break the run nor mask the +thing being checked — which is also why the old "or it is an unrelated Iris +deprecation, note it and move on" escape hatch has gone.) + +- [ ] **Step 17: Green check and commit** + +```bash +pre-commit run --files lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py +pytest lib/iris/tests/unit/fileformats/netcdf/dataset/ +git add lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py +git commit -m "$(cat <<'EOF' +Deprecate reaching netCDF4 API through a CFVariable + +CFVariable.__getattr__ falls back to the backing variable for any name +the CF attributes do not supply. That was how the class worked, and +third-party code will have relied on it, so it stays for one release +cycle - but it now says so. + +The warning waits until here because it could not have been switched on +earlier: consumers came off the fallback one task at a time, and a +warning during that would have fired once per attribute per variable per +load. Nothing in Iris takes the route now, which the -W error run over +the netCDF, CF and integration suites demonstrates. + +It warns only on success. hasattr() probes reach this method, and a +probe that comes back False has not used anything. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 12: `netcdf/saver.py` — `Saver` builds and uses a `NetCDFDataset` + +The last consumer, and the only one that writes. Everything the saver does +to a file goes through four verbs — declare a dimension, create a variable, +set an attribute, put data in — and this task routes all four through +`CFDataset`. After it, `saver.py` names `netCDF4` in exactly one place: the +`self._dataset.dataset` escape hatch, kept deliberately and **three times** +only. (Corrected in Task 13, which counted them: the `file_format` read +that decides whether int64 is available, the second `file_format` read in +the error message it raises, and the third-party `cf_patch` hook, +documented since Iris 1.3 as receiving netCDF4 objects. The step text below +already says three; only this summary said two.) + +**Files:** +- Modify: `lib/iris/fileformats/netcdf/saver.py` (throughout; sites listed per step) +- Modify: `lib/iris/fileformats/netcdf/_dataset.py` (`NetCDFDataset.closed`) +- Test: `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py` (create) +- Test: `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py` (create) +- Test: `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py` (mocks and patch targets) +- Test: `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py` (mocks) + +**Interfaces:** +- Consumes: `NetCDFDataset`, `NetCDFDatasetVariable` and `_bytes_if_ascii` + from Tasks 2–4; `NetCDFDataset.write_lock`; `NetCDFDatasetVariable.attributes`, + `.write_handle()`, `.is_emulated`, `.emulated_data_array`, `.variable`. +- Produces: + - `Saver._dataset` is a `NetCDFDataset`, not an `EncodedDataset`. Anything + reaching for netCDF4 API on it must go via `Saver._dataset.dataset`. + - `Saver.file_write_lock` is `Saver._dataset.write_lock` — the *same* + object every `NetCDFDatasetVariable` of that dataset holds. + - `saver._setncattr` is **deleted**. `saver._bytes_if_ascii` was already + deleted in Task 4; its re-import goes too. + - `saver.CFVariable` (the type alias, not `cf.CFVariable`) becomes + `NetCDFDatasetVariable`. + - `NetCDFDataset.closed` reports a dataset closed behind its back. + +> **One long red stretch.** Steps 5–10 convert the saver. There is no useful +> place to stop in the middle: the saver creates variables in one method and +> writes attributes to them in another, so a half-converted saver cannot write +> a file at all. Steps 5–10 therefore run with no test run between them, and +> the green check is Step 11. This is the second and last such stretch in the +> plan; Task 11's note explains why they are worth one commit each. + +- [ ] **Step 1: Write the Review Focus 4 test — saving into a dataset somebody else opened** + +This is a **characterisation** test: it passes against today's code, before +anything in this task changes. `iris.fileformats.netcdf.save(cube, dataset)` +— the Xarray bridge, [#4994](https://github.com/SciTools/iris/issues/4994), +and the route [ncdata](https://github.com/pp-mo/ncdata) takes — has no test +anywhere in the suite today, and this task rewrites every line it touches. +Writing the guard first is the only way the conversion has anything to break. + +Create +`lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Tests for saving into a dataset the caller opened, or is only pretending to. + +``iris.fileformats.netcdf.save(cube, dataset, compute=False)`` accepts an +open dataset in place of a path. Two kinds of caller do this: + +* one holding a real, open netCDF4 dataset, who wants Iris to add to it; +* one holding an object that merely *looks* like a netCDF4 dataset, and + collects what Iris writes rather than storing it - the "Xarray bridge", + https://github.com/SciTools/iris/issues/4994, which is how + https://github.com/pp-mo/ncdata translates between Iris and Xarray. + +The second is the reason ``Saver`` may not assume its dataset is netCDF4, +and it is entirely untested elsewhere. + +""" + +import dask +import numpy as np +import pytest + +import iris +from iris.coords import DimCoord +from iris.cube import Cube +import iris.fileformats.netcdf as inetcdf +from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc + + +@pytest.fixture(autouse=True) +def split_attrs(): + # Avoid the legacy-attribute deprecation warning; irrelevant here. + with iris.FUTURE.context(save_split_attrs=True): + yield + + +@pytest.fixture +def cube(): + cube = Cube( + np.arange(6.0).reshape(2, 3), + standard_name="air_temperature", + units="K", + var_name="air", + ) + cube.add_dim_coord( + DimCoord(np.arange(2.0), standard_name="latitude", units="degrees"), 0 + ) + cube.add_dim_coord( + DimCoord(np.arange(3.0), standard_name="longitude", units="degrees"), 1 + ) + return cube + + +class TestRealDataset: + """A dataset the caller opened, and closes themselves.""" + + def test_save_into_an_open_dataset(self, cube, tmp_path): + path = tmp_path / "user.nc" + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + delayed = inetcdf.save(cube, dataset, compute=False) + # The caller owns the dataset, so the caller closes it - and the + # delayed writes only resolve once they have. + dataset.close() + dask.compute(delayed) + + result = iris.load_cube(path) + assert result.standard_name == "air_temperature" + assert result.units == "K" + assert np.array_equal(result.data, cube.data) + assert [coord.name() for coord in result.coords()] == [ + "latitude", + "longitude", + ] + + def test_saver_does_not_close_what_it_was_given(self, cube, tmp_path): + path = tmp_path / "user.nc" + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + inetcdf.save(cube, dataset, compute=False) + assert dataset.isopen() + dataset.close() + + def test_compute_true_is_refused(self, cube, tmp_path): + path = tmp_path / "user.nc" + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + with pytest.raises(ValueError, match="Cannot save to a user-provided dataset"): + inetcdf.save(cube, dataset, compute=True) + dataset.close() + + +class _EmulatedDimension: + def __init__(self, size): + self._size = size + + def __len__(self): + return self._size + + def isunlimited(self): + return self._size is None + + +class _EmulatedVariable: + """The least a netCDF4.Variable emulator can be and still be written to. + + ``_data_array`` is the whole point: an emulator receives its data by + having this attribute set, never by ``__setitem__``. ``__setitem__`` + raises here so that a regression shows up as a failure rather than as + data quietly going nowhere. + + """ + + def __init__(self, name, datatype, dimensions, shape): + self.name = name + self.datatype = np.dtype(datatype) + self.dtype = self.datatype + self.dimensions = tuple(dimensions) + self.shape = shape + self.size = int(np.prod(shape)) if shape else 1 + self._data_array = None + self._attrs = {} + + def setncattr(self, name, value): + self._attrs[name] = value + + def getncattr(self, name): + return self._attrs[name] + + def ncattrs(self): + return list(self._attrs) + + def chunking(self): + return "contiguous" + + def __setitem__(self, keys, values): + raise AssertionError("An emulated variable must receive _data_array.") + + +class _EmulatedDataset: + """A netCDF4.Dataset emulator, in the shape ncdata presents. + + Deliberately *not* a ``_thread_safe_nc`` wrapper and deliberately without + ``THREAD_SAFE_FLAG``, so that the wrapping branch of + ``NetCDFDataset.from_existing`` is the one under test. + + """ + + def __init__(self): + self.variables = {} + self.dimensions = {} + self.file_format = "NETCDF4" + self._attrs = {} + self._open = True + + def createDimension(self, name, size): + self.dimensions[name] = _EmulatedDimension(size) + return self.dimensions[name] + + def createVariable(self, name, datatype, dimensions=(), **kwargs): + shape = tuple(len(self.dimensions[name_]) for name_ in dimensions) + variable = _EmulatedVariable(name, datatype, dimensions, shape) + self.variables[name] = variable + return variable + + def setncattr(self, name, value): + self._attrs[name] = value + + def getncattr(self, name): + return self._attrs[name] + + def ncattrs(self): + return list(self._attrs) + + def sync(self): + pass + + def close(self): + self._open = False + + def isopen(self): + return self._open + + def filepath(self): + return "" + + def set_auto_chartostring(self, onoff): + pass + + +class TestEmulatedDataset: + """An object that only looks like a dataset - the Xarray bridge.""" + + @pytest.fixture + def written(self, cube): + dataset = _EmulatedDataset() + inetcdf.save(cube, dataset, compute=False) + return dataset + + def test_variables_are_created(self, written): + assert sorted(written.variables) == ["air", "latitude", "longitude"] + + def test_dimensions_are_created(self, written): + assert {name: len(dim) for name, dim in written.dimensions.items()} == { + "latitude": 2, + "longitude": 3, + } + + def test_attributes_reach_the_emulator_as_bytes(self, written): + # _bytes_if_ascii: an ASCII string attribute is offered as bytes, so + # that netCDF4 gives it type NC_CHAR. An emulator sees the same. + assert written.variables["air"]._attrs == { + "standard_name": b"air_temperature", + "units": b"K", + } + assert written._attrs == {"Conventions": b"CF-1.7"} + + def test_data_arrives_as_a_data_array(self, written, cube): + # Not via __setitem__, which the emulated variable refuses. + assert np.array_equal(written.variables["air"]._data_array, cube.data) + assert np.array_equal( + written.variables["latitude"]._data_array, np.arange(2.0) + ) + + def test_dimensions_are_recorded_on_the_variable(self, written): + assert written.variables["air"].dimensions == ("latitude", "longitude") +``` + +- [ ] **Step 2: Run the test to verify it passes** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py -v` +Expected: PASS — all 8 tests, against unmodified `saver.py`. + +A green test at the start of a task is unusual, and is the point here: from +Step 5 the saver stops touching netCDF4 directly, and this file is the only +thing in the suite that would notice if the emulated path broke. Confirm it +really bites before relying on it: + +```bash +python - <<'PY' +import pathlib +p = pathlib.Path("lib/iris/fileformats/netcdf/saver.py") +src = p.read_text() +p.write_text(src.replace('if hasattr(cf_var, "_data_array"):', "if False:", 1)) +PY +pytest lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py \ + -k Emulated -q +git checkout lib/iris/fileformats/netcdf/saver.py +``` +Expected: the four `TestEmulatedDataset` data/attribute tests FAIL while the +line is disabled, and the file is restored afterwards. + +- [ ] **Step 3: Commit the guard** + +```bash +pre-commit run --files \ + lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py +git add lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py +git commit -m "$(cat <<'EOF' +Characterise saving into a caller-supplied dataset + +iris.fileformats.netcdf.save() accepts an open dataset in place of a +path, and ncdata uses that to translate between Iris and Xarray by +passing an object that only emulates one. Nothing in the suite tested +it, and the next commit rewrites every line it runs through. + +Written against unmodified code, so it is green on arrival. That is +what makes it a guard rather than a specification. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 4: Write the failing tests for the dataset swap** + +Create +`lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py`: + +```python +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Tests for the dataset :class:`iris.fileformats.netcdf.saver.Saver` owns.""" + +import numpy as np +import pytest + +from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc +from iris.fileformats.netcdf._dataset import NetCDFDataset +from iris.fileformats.netcdf.saver import Saver + + +class TestOwnedDataset: + @pytest.fixture + def saver(self, tmp_path): + with Saver(tmp_path / "test.nc", "NETCDF4") as saver: + yield saver + + def test_dataset_is_a_netcdf_cfdataset(self, saver): + assert isinstance(saver._dataset, NetCDFDataset) + + def test_filepath_is_still_a_path(self, saver, tmp_path): + # Public API: Saver.filepath has always been a Path for a local file. + assert saver.filepath == (tmp_path / "test.nc").absolute() + + def test_the_write_lock_is_the_datasets_own(self, saver): + # Every variable of the dataset holds this same lock. A second lock + # would exclude nothing - see NetCDFDataset.write_lock. + assert saver.file_write_lock is saver._dataset.write_lock + + def test_format_reaches_the_file(self, tmp_path): + with Saver(tmp_path / "classic.nc", "NETCDF3_CLASSIC") as saver: + assert saver._dataset.dataset.file_format == "NETCDF3_CLASSIC" + + def test_exit_closes_an_owned_dataset(self, tmp_path): + with Saver(tmp_path / "test.nc", "NETCDF4") as saver: + pass + assert saver._dataset.closed + + +class TestBorrowedDataset: + @pytest.fixture + def raw(self, tmp_path): + dataset = threadsafe_nc.DatasetWrapper( + tmp_path / "user.nc", mode="w", format="NETCDF4" + ) + yield dataset + if dataset.isopen(): + dataset.close() + + def test_dataset_is_a_netcdf_cfdataset(self, raw): + with Saver(raw, "NETCDF4", compute=False) as saver: + assert isinstance(saver._dataset, NetCDFDataset) + + def test_exit_leaves_a_borrowed_dataset_open(self, raw): + with Saver(raw, "NETCDF4", compute=False) as saver: + pass + assert raw.isopen() + assert not saver._dataset.closed + + def test_complete_refuses_while_the_file_is_open(self, raw): + with Saver(raw, "NETCDF4", compute=False) as saver: + pass + with pytest.raises(ValueError, match="until its dataset is closed"): + saver.complete() + + def test_complete_sees_a_dataset_closed_behind_its_back(self, raw): + # The caller owns the dataset and closes it themselves, so the flag + # NetCDFDataset.close() sets is never set. complete() has to ask the + # file, not the flag. + with Saver(raw, "NETCDF4", compute=False) as saver: + pass + raw.close() + assert saver._dataset.closed + saver.complete() # must not raise + + +class TestWritingThroughTheDataset: + """The four verbs, through CFDataset rather than netCDF4.""" + + @pytest.fixture + def saver(self, tmp_path): + with Saver(tmp_path / "test.nc", "NETCDF4") as saver: + yield saver + + def test_create_dimension(self, saver): + saver._dataset.create_dimension("x", 3) + assert saver._dataset.dimensions["x"] == 3 + + def test_create_variable_returns_a_cf_dataset_variable(self, saver): + from iris.fileformats.netcdf._dataset import NetCDFDatasetVariable + + saver._dataset.create_dimension("x", 3) + variable = saver._dataset.create_variable("a", np.dtype("f4"), ("x",)) + assert isinstance(variable, NetCDFDatasetVariable) + assert saver._dataset.variables["a"] is variable +``` + +- [ ] **Step 5: Run the tests to verify they fail** + +Run: `pytest lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py -v` +Expected: FAIL — `assert isinstance(saver._dataset, NetCDFDataset)` fails +(it is an `EncodedDataset`), `saver._dataset.write_lock` raises +`AttributeError`, and `saver._dataset.closed` raises `AttributeError`. + +- [ ] **Step 6: Build a `NetCDFDataset` in `Saver.__init__`, and use it in `__exit__` and `complete`** + +In `lib/iris/fileformats/netcdf/saver.py`, add the import alongside the +existing netCDF imports at line 65: + +```python +from iris.fileformats.netcdf._dataset import NetCDFDataset, NetCDFDatasetVariable +``` + +Replace lines 425-480 (`self._to_open_dataset = ...` down to +`self.file_write_lock = ...`) with: + +```python + # Detect if we were passed a pre-opened dataset (or something like one) + self._to_open_dataset = hasattr(filename, "createVariable") + if self._to_open_dataset: + # We were passed a *dataset*, so we don't open (or close) one of our own. + if compute: + msg = ( + "Cannot save to a user-provided dataset with 'compute=True'. " + "Please use 'compute=False' and complete delayed saving in the " + "calling code after the file is closed." + ) + raise ValueError(msg) + + # from_existing() handles the thread-safety wrapping, including the + # case of an object that only emulates a dataset and carries no + # wrapper of its own. + self._dataset = NetCDFDataset.from_existing(filename) + + # In this case the dataset gives a filepath, not the other way around. + self.filepath = self._dataset.location + + else: + # Given a filepath string/path : create a dataset from that + try: + # Lazy import to avoid circular import overhead at module import-time. + from iris.io import _is_nczarr_fragment + + self._is_nczarr = _is_nczarr_fragment( + urlsplit(str(filename)).fragment or None + ) + if self._is_nczarr: + # NCZarr URLs contain a #mode= fragment; Path() strips it. + # Keep as a plain string and pass directly to the dataset. + self.filepath = str(filename) + else: + filepath = Path(filename) + self.filepath = filepath.absolute() + self._dataset = NetCDFDataset( + self.filepath, mode="w", netcdf_format=netcdf_format + ) + except RuntimeError: + if self._is_nczarr: + raise + dir_name = Path(self.filepath).parent + if not dir_name.is_dir(): + msg = "No such file or directory: {}".format(dir_name) + raise IOError(msg) + if not os.access(dir_name, os.R_OK | os.W_OK): + msg = "Permission denied: {}".format(self.filepath) + raise IOError(msg) + else: + raise + + # One lock for the whole file, shared with every variable of it. A + # second get_worker_lock() call would hand back a different + # threading.Lock under the threaded scheduler, excluding nothing. + self.file_write_lock = self._dataset.write_lock +``` + +`Saver.filepath` stays exactly what it was — a `Path` for a local file, the +URL string for NCZarr, whatever the dataset reports for a borrowed one. +`NetCDFDataset` keeps its own `str()` of it internally; the two are not the +same object and are not required to be. + +Replace `__exit__` (lines 485-497): + +```python + def __exit__(self, type, value, traceback): + """Flush any buffered data to the CF-NetCDF/NcZarr file before closing.""" + if self._nczarr_writes: + # NCZarr lazy writes must be computed while the file is still open; + # the deferred reopen-write pattern used for netCDF is not supported. + sources, targets = zip(*self._nczarr_writes) + da.store(list(sources), list(targets)) + self._dataset.sync() + if not self._to_open_dataset: + # Only close if the Saver created it. + self._dataset.close() + # Complete after closing, if required + if self.compute: + self.complete() +``` + +`finalise()` is deliberately **not** called here. It exists on the ABC from +the start because spec §4.5 requires it, and for netCDF it does nothing; +wiring a no-op in would be a guess at an ordering only Zarr can exercise. +PR 6 wires it, where it does something. See finding F6. + +And in `complete()` (line 2694), replace `if self._dataset.isopen():` with: + +```python + if not self._dataset.closed: +``` + +- [ ] **Step 7: Teach `NetCDFDataset.closed` about a dataset closed behind its back** + +`Saver.complete()` is the first caller for which the flag alone is not the +answer: the documented workflow for a caller-supplied dataset is +`compute=False`, *caller* closes, `complete()`. `NetCDFDataset.close()` never +runs in that sequence, because the dataset is borrowed. + +In `lib/iris/fileformats/netcdf/_dataset.py`, replace the `closed` property: + +```python + @property + def closed(self) -> bool: + """Whether the file has been released - by this object or by its owner. + + The flag :meth:`close` sets is not the whole answer for a borrowed + dataset, which whoever opened it closes themselves. Ask the backing + object as well, when it can say: an emulating object need not + implement ``isopen()``, and is then taken to be open. + + """ + if self._closed: + return True + isopen = getattr(self._dataset, "isopen", None) + if isopen is None: + return False + return not isopen() +``` + +- [ ] **Step 8: Move dimension and variable creation onto the dataset** + +Four `createDimension` calls become `create_dimension`, at lines 833, 929, +1588 and 1930: + +```python + self._dataset.create_dimension(dim_name, size) # 833 + self._dataset.create_dimension(last_dim, length) # 929 + self._dataset.create_dimension( # 1588 + bounds_dimension_name, n_bounds + ) + self._dataset.create_dimension( # 1930 + string_dimension_name, string_dimension_depth + ) +``` + +Five `createVariable` calls become `create_variable`, at lines 1592, 1720, +1938, 1983 and 2080: + +```python + cf_var_bounds = self._dataset.create_variable( # 1592 + boundsvar_name, + bounds.dtype.newbyteorder("="), + cf_var.dimensions + (bounds_dimension_name,), + **compression_kwargs, + ) + + # Create the main variable + cf_mesh_var = self._dataset.create_variable( # 1720 + cf_mesh_name, + np.dtype(np.int32), + ) + # 'dimensions' defaults to (), so the empty list goes - finding F4. + + cf_var = self._dataset.create_variable( # 1938 + cf_name, "|S1", element_dims + ) + + cf_var = self._dataset.create_variable( # 1983 + cf_name, + dtype, + element_dims, + fill_value=fill_value, + **compression_kwargs, + ) + + cf_var_grid = self._dataset.create_variable( # 2080 + cs.grid_mapping_name, np.int32 + ) +``` + +`create_variable` normalises `dimensions` with `tuple()`, so the lists the +saver passes (`element_dims`, `["dim0", "dim1"]`) reach netCDF4 as tuples. +That is invisible to netCDF4 and visible to two mock-based tests, which +Step 11 updates. + +Then the grid-mapping escape hatch. `_add_grid_mapping_to_dataset` +(lines 2067-2272) sets sixty-three CF grid-mapping parameters by plain Python +attribute assignment, one per CF parameter name. They bypass +`_bytes_if_ascii` today — `crs_wkt` at line 2273 is written as `NC_STRING`, +not `NC_CHAR` — so converting them to `attributes[...]` entries would change +what lands in the file. That is finding F8, and it is not this PR's job. +Keep them on the netCDF4 variable, named once: + +```python + cf_var_grid = self._dataset.create_variable( + cs.grid_mapping_name, np.int32 + ) + cf_var_grid.attributes["grid_mapping_name"] = cs.grid_mapping_name + + # The sixty-three assignments below set CF grid-mapping parameters by + # Python attribute assignment. Unlike every other attribute the saver + # writes, they bypass the ASCII-to-bytes coercion, so moving them onto + # .attributes would change the file. See finding F8; until then they + # need the netCDF4 variable itself. + grid_variable = cf_var_grid.variable +``` + +and rename the rest mechanically: + +```bash +python - <<'PY' +import pathlib +p = pathlib.Path("lib/iris/fileformats/netcdf/saver.py") +lines = p.read_text().splitlines(keepends=True) +start = next( + i for i, line in enumerate(lines) if "grid_variable = cf_var_grid.variable" in line +) +end = next( + i for i, line in enumerate(lines) if "def _create_cf_grid_mapping" in line +) +for i in range(start + 1, end): + lines[i] = lines[i].replace("cf_var_grid.", "grid_variable.") +p.write_text("".join(lines)) +print("renamed in lines", start + 2, "to", end) +PY +``` +Expected: `renamed in lines to `, and +`grep -c "cf_var_grid\." lib/iris/fileformats/netcdf/saver.py` reports `2`. + +- [ ] **Step 9: Move every attribute write onto `.attributes`** + +Thirty-three of the thirty-four `_setncattr` calls are a single line of the +form `_setncattr(target, name, value)`. Rewrite them by splitting the +argument list on top-level commas, so that values containing commas inside +their own brackets survive: + +```bash +python - <<'PY' +import pathlib +import re + +SRC = pathlib.Path("lib/iris/fileformats/netcdf/saver.py") +lines = SRC.read_text().splitlines(keepends=True) + + +def split_args(text): + """Split a call's argument text on top-level commas.""" + args, depth, start = [], 0, 0 + for i, ch in enumerate(text): + if ch in "([{": + depth += 1 + elif ch in ")]}": + depth -= 1 + elif ch == "," and depth == 0: + args.append(text[start:i].strip()) + start = i + 1 + args.append(text[start:].strip()) + return args + + +out, changed = [], 0 +for line in lines: + match = re.match(r"^(\s*)_setncattr\((.*)\)\s*$", line.rstrip("\n")) + if match and match.group(2): + indent, inner = match.groups() + args = split_args(inner) + if len(args) == 3: + target, name, value = args + out.append(f"{indent}{target}.attributes[{name}] = {value}\n") + changed += 1 + continue + out.append(line) + +SRC.write_text("".join(out)) +print("rewritten:", changed) +PY +``` +Expected: `rewritten: 33`. + +Then the one multi-line call, at line 1728: + +```python + cf_mesh_var.attributes["topology_dimension"] = np.int32( + mesh.topology_dimension + ) +``` + +Then delete `_setncattr` itself (lines 290-301) and the `_bytes_if_ascii` +import Task 4 left behind. Nothing in Iris calls either any more: the +coercion now happens inside `_NetCDFAttributes.__setitem__`, once, for every +attribute the saver writes. + +Three attribute *reads* on netCDF variables also have to move, and they are +the ones that would fail silently if missed. `NetCDFDatasetVariable` has no +`__getattr__` fallback — deliberately, see `lib/iris/AGENTS.md` — so +`hasattr(cf_var, "formula_terms")` would simply become permanently `False` +and the dimensionless-vertical-coordinate branch would stop running. + +At line 1182, in `_add_aux_factories`: + +```python + if "formula_terms" in cf_var.attributes: + if ( + cf_var.attributes["formula_terms"] != formula_terms + or cf_var.attributes["standard_name"] != std_name + ): +``` + +At lines 1226-1240, in the same method's `FUTURE.derived_bounds` block: + +```python + bounds_varname = cf_var.attributes.get("bounds") + cf_bounds_var = self._dataset.variables.get(bounds_varname, None) + if ( + cf_bounds_var is not None + and cf_bounds_var.attributes.get("formula_terms") is None + ): +``` + +and, inside `boundsterm_varname`: + +```python + termvar = self._dataset.variables.get(term_varname) + boundsname = termvar.attributes.get("bounds") +``` + +`termvar` cannot be `None` here: `term_varname` comes from +`self._name_coord_map`, which only holds names the saver has created. +`getattr(None, "bounds", None)` would have returned `None` silently; if the +assumption is ever wrong an `AttributeError` now says so. + +At line 1802, in `_set_cf_var_attributes`: + +```python + # Don't clobber existing attributes. + if name not in cf_var.attributes: + cf_var.attributes[name] = value +``` + +This one is a **deliberate behaviour change**, and the saver-side mirror of +Review Focus 1. `hasattr(cf_var, name)` asked a netCDF4 variable whether it +had a Python member of that name, so a coordinate carrying an attribute +called `shape`, `size`, `name`, `dtype`, `dimensions` or `mask` was silently +dropped: `hasattr` said yes, and the value never reached the file. netCDF +itself has no such restriction — `setncattr("shape", ...)` and +`getncattr("shape")` both work, and the array shape is still reported +correctly — so the attribute now round-trips. Finding F13. + +Two netCDF-only reads stay, through the escape hatch. `_ensure_valid_dtype` +needs the file format to know whether int64 is available, at lines 1516 and +1536: + +```python + ) and self._dataset.dataset.file_format in ( + msg = msg.format(src_name, src_object, self._dataset.dataset.file_format) +``` + +and `cf_patch`, at line 733, is a third-party hook documented as receiving a +netCDF4 dataset: + +```python + cf_patch(profile, self._dataset.dataset, cf_var_cube) +``` + +`self._dataset.dataset` is deliberately ugly. Grepping for it finds every +place `saver.py` still assumes netCDF4, and there are exactly three. + +- [ ] **Step 10: Route data writes through the variable** + +Replace `_lazy_stream_data` (lines 2598-2656): + +```python + def _lazy_stream_data( + self, + data: np.typing.ArrayLike, + cf_var: NetCDFDatasetVariable, + ) -> None: + if hasattr(data, "shape") and data.shape == (1,) + cf_var.shape: + # (Don't do this check for string data). + # Reduce dimensionality where the data array has an extra dimension + # versus the cf_var - to avoid a broadcasting ambiguity. + # Happens when bounds data is for a scalar point - array is 2D but + # contains just 1 row, so the cf_var is 1D. + data = data.squeeze(axis=0) + + if cf_var.is_emulated: + # The variable is not an actual file variable, but an emulating + # object with an attached data array (either numpy or dask), which should + # be copied immediately to the target. This is used as a hook to translate + # data to/from netcdf data container objects in other packages, such as + # xarray. + # See https://github.com/SciTools/iris/issues/4994 "Xarray bridge". + cf_var.emulated_data_array = data + + else: + doing_delayed_save = is_lazy_data(data) + if doing_delayed_save: + if self._is_nczarr: + # NCZarr cannot be reopened for deferred writes. + # Collect here; da.store() will be called in __exit__. + self._nczarr_writes.append((data, cf_var)) + return + + # save lazy data with a delayed operation. For now, we just record the + # necessary information -- a single, complete delayed action is + # constructed later by a call to delayed_completion(). + def store( + data: np.typing.ArrayLike, + cf_var: NetCDFDatasetVariable, + ) -> None: + # Ask the variable for something a worker can stream into + # after this file is closed. What that is is the backend's + # business: netCDF reopens the file, Zarr will hand back the + # array itself. + self._delayed_writes.append((data, cf_var.write_handle())) + + else: + # Real data is always written directly, i.e. not via lazy save. + def store( + data: np.typing.ArrayLike, + cf_var: NetCDFDatasetVariable, + ) -> None: + cf_var[:] = data # type: ignore[index] + + # Store the data. + store(data, cf_var) +``` + +The write proxy was built here from `self.filepath`, `cf_var` and +`self.file_write_lock`; the variable now supplies all three itself, and +`file_write_lock` is the dataset's own lock, so the two are the same object +rather than two locks that happened to agree. + +Then the type alias at line 326, and the import at line 67: + +```python +# A saver variable is whatever the dataset hands back. For netCDF that is a +# NetCDFDatasetVariable, whether it wraps a real file variable or an +# emulating object - see VariableEmulator, above. +CFVariable = NetCDFDatasetVariable +``` + +`from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc` was +used only by the three type hints just replaced: delete it. +`bytecoding_datasets` stays — lines 1916 and 1919 still need +`_identify_encoding` and `_ENCODING_WIDTH_TRANSLATIONS`. + +- [ ] **Step 11: Update the mocks that stood in for netCDF4 objects** + +Three existing test modules build mocks shaped like netCDF4 objects, and now +need mocks shaped like `CFDataset` ones. + +`lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py`, `test_zlib` +(lines 218-246). It patched the module `saver.py` opened its dataset from; +`saver.py` no longer opens one, so patch the module that does, and expect a +tuple of dimension names: + +```python + def test_zlib(self, mocker): + cube = self._simple_cube(">f4") + api = mocker.patch( + "iris.fileformats.netcdf._dataset._bytecoding_datasets" + ) + # Mock the apparent dtype of mocked variables, to avoid an error. + dataset = api.EncodedDataset.return_value + dataset.createVariable.return_value.dtype = np.dtype(np.float32) + # NOTE: use compute=False as otherwise it gets in a pickle trying to construct + # a fill-value report on a non-compliant variable in a non-file (!) + with Saver("/dummy/path", "NETCDF4", compute=False) as saver: + saver.write(cube, zlib=True) + create_var_call = mocker.call( + "air_pressure_anomaly", + np.dtype("float32"), + ("dim0", "dim1"), + fill_value=None, + shuffle=True, + least_significant_digit=None, + contiguous=False, + zlib=True, + fletcher32=False, + endian="native", + complevel=4, + chunksizes=None, + ) + assert create_var_call in dataset.createVariable.call_args_list +``` + +The `api.default_fillvals` line goes with it: `saver.py` never read +`bytecoding_datasets.default_fillvals`, so the line has been inert since the +netCDF modules were split, and it now patches a module the saver does not +touch at all. + +The three `createvar_spy` blocks in the same file (lines 273-279, 313-319, +353-359) keep their patch target — `NetCDFDataset.create_variable` calls +`EncodedDataset.createVariable`, so the spy still sees every call — but +`wraps=` must name the object that now has that method: + +```python + wraps=saver._dataset.dataset.createVariable, +``` + +`_check_bounds_setting` (lines 499-527) builds a variable mock and asserts on +`setncattr`: + +```python + var = self.mocker.MagicMock(spec=NetCDFDatasetVariable) + var.attributes = {} + var.dimensions = ("time",) + + # Make the main call. + Saver._create_cf_bounds(saver, coord, var, "time") + + # Test the attribute written by _create_cf_bounds. The ASCII-to-bytes + # coercion now happens inside _NetCDFAttributes.__setitem__, and is + # tested there; this plain dict stands in for one, so the value + # arrives as given. + assert var.attributes[property_name] == boundsvar_name + + # Test the call of create_variable in _create_cf_bounds. + dataset = saver._dataset + expected_dimensions = var.dimensions + ("bnds",) + create_var_call = self.mocker.call( + boundsvar_name, coord.bounds.dtype, expected_dimensions + ) + assert create_var_call == dataset.create_variable.call_args +``` + +with `from iris.fileformats.netcdf._dataset import NetCDFDatasetVariable` +added to the imports, and the now-unused `ds_wrappers.EncodedVariable` +reference removed if nothing else in the file needs it. + +`lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py`, +`mock_var` (lines 61-80) and the assertions that use it: + +```python + @staticmethod + def mock_var(shape, with_data_array, mocker): + # Create a test cf_var object. + # 'is_emulated' is now a declared property rather than the presence of + # a '_data_array' member, so it can simply be set. + mock_cfvar = mocker.MagicMock( + spec=dataset_module.NetCDFDatasetVariable, + shape=tuple(shape), + dtype=np.dtype(np.float32), + is_emulated=with_data_array, + ) + mock_cfvar.write_handle.return_value = mocker.sentinel.write_handle + # Give the mock cf-var a name property, as required by '_lazy_stream_data'. + # This *can't* be an extra kwarg to MagicMock __init__, since that already + # defines a specific 'name' kwarg, with a different purpose. + mock_cfvar.name = "" + return mock_cfvar +``` + +with `import iris.fileformats.netcdf._dataset as dataset_module` added, and +in `test_data_save`: + +```python + if data_form == "lazydata": + result_data, result_writer = saver._delayed_writes[0] + assert result_data is data + # What kind of handle it is is the dataset's business, and is + # tested in test_NetCDFDataset__write.py. + assert result_writer is mocker.sentinel.write_handle + elif data_form == "realdata": + cf_var.__setitem__.assert_called_once_with(slice(None), data) + else: + assert data_form == "emulateddata" + assert cf_var.emulated_data_array is data +``` + +Note the last line: the original wrote `cf_var._data_array == ...` with no +`assert`, so it asserted nothing at all. + +- [ ] **Step 12: Run the netCDF suites to verify they pass** + +Run: +```bash +pytest lib/iris/tests/unit/fileformats/netcdf/ \ + lib/iris/tests/unit/fileformats/cf/ \ + lib/iris/tests/integration/netcdf/ +``` +Expected: PASS — `0 failed, 23 skipped, 0 errors`. That is 1 + 1 + 21 from +§6.2's per-selection table; run serially like this, F14 does not fire, so +there is nothing to excuse here. A new skip is a test that stopped running, +not a test that passed. + +`test_Saver__user_dataset.py` passing here is the point of Step 1: the +emulated-dataset path went through `NetCDFDataset.from_existing`, +`create_dimension`, `create_variable`, `_NetCDFAttributes.__setitem__` and +`emulated_data_array` on this run, and none of them existed when it was +written. + +- [ ] **Step 13: Green check and commit** + +```bash +pytest -n auto lib/iris/tests/unit/ lib/iris/tests/integration/ +pre-commit run --files lib/iris/fileformats/netcdf/saver.py \ + lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py \ + lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py +git add lib/iris/fileformats/netcdf/saver.py \ + lib/iris/fileformats/netcdf/_dataset.py \ + lib/iris/tests/unit/fileformats/netcdf/saver/ +git commit -m "$(cat <<'EOF' +Save through CFDataset rather than through netCDF4 + +Saver now owns a NetCDFDataset. Dimensions, variables, attributes and +data writes all go through the interface, so the saver no longer knows +what kind of file it is writing. + +Three netCDF-only reads remain, all through the explicit +'self._dataset.dataset' escape hatch: the file-format check that decides +whether int64 is available, and the third-party cf_patch hook, which is +documented as receiving a netCDF4 dataset. Grep for it to find them. + +The grid-mapping variable keeps a netCDF4 handle too. Its sixty-three +parameter assignments bypass the ASCII-to-bytes coercion every other +attribute goes through, so converting them would change the file - a +separate change, recorded as finding F8. + +One deliberate behaviour change: a coordinate attribute whose name +collides with a netCDF4 Python member - 'shape', 'size', 'dtype' and the +rest - used to be dropped silently, because the "don't clobber" check +asked hasattr(). It asks the attribute mapping now, and the attribute +reaches the file. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +--- + +### Task 13: Spec updates, the full-suite diff, and the pull request + +The interface in `cf/dataset.py` is now the real one, and the sketch in spec +§4.2 is a year-zero guess that six findings have refined. They have to agree, +because PRs 3–6 are written against the spec. Then the claim this whole plan +rests on — "no behaviour change other than the two named ones" — gets +demonstrated rather than asserted, and the pull request goes up. + +**Files:** +- Modify: `docs/superpowers/specs/2026-09-21-zarr-io-design.md` (§4.2, §4.3, §12.1, §12.3, §12.4, §12.6) +- Test: none. This task changes no code. + +**Interfaces:** +- Consumes: the interface as built in Tasks 1–4, and findings F1–F13. +- Produces: a spec whose §4.2 matches `cf/dataset.py`, and an open pull + request against `SciTools/iris:brownfield`. + +- [ ] **Step 1: Bring spec §4.2 into line with the code** + +In `docs/superpowers/specs/2026-09-21-zarr-io-design.md`, replace the code +block in §4.2 with what Tasks 1–4 actually built: + +```python +class CFDatasetVariable(ABC): + """One named array in a CF-conforming dataset.""" + name: str + location: str # the dataset's path or URL; F2 + dimensions: tuple[str, ...] + shape: tuple[int, ...] + dtype: np.dtype + size: int + fill_value: Any | None + chunking: tuple[int, ...] | None # None when the store is unchunked + attributes: MutableMapping[str, Any] # materialised once + + def __getitem__(self, keys) -> np.ndarray: ... + def __setitem__(self, keys, values) -> None: ... + def __len__(self) -> int: ... # defaults to self.shape[0]; F5 + + def write_handle(self) -> Any: ... # picklable __setitem__ target; see 4.5 + def deprecated_netcdf_member(self, name: str) -> Any: ... # see 4.3 + + +class CFDataset(ABC): + """A CF-conforming array store, open for reading or writing.""" + location: str # path or URL, for messages and proxies + mode: str # "r" | "r+" | "a" | "w" | "w-"; 4.5 + closed: bool # F3 + variables: Mapping[str, CFDatasetVariable] + dimensions: Mapping[str, int] + attributes: MutableMapping[str, Any] + + def create_dimension(self, name: str, size: int | None) -> None: ... + def create_variable(self, name, dtype, dimensions=(), *, + fill_value=None, **encoding) -> CFDatasetVariable: ... + def sync(self) -> None: ... + def finalise(self) -> None: ... # one-shot; NOT part of close() + def close(self) -> None: ... +``` + +and replace the paragraph beginning "Three members exist only to keep +multi-process writing reachable later" with: + +```markdown +Three members exist only to keep multi-process writing reachable later, and +are explained in §4.5: `write_handle`, `mode` and `finalise`. They cost +almost nothing now — `mode` is a string the implementations already track +internally, `write_handle` returns `self` for Zarr, and `finalise` is a no-op +for netCDF — and their absence is what would force a breaking interface +change later. `finalise` is defined but not yet called: PR 2 left +`Saver.__exit__` alone rather than guess at an ordering only Zarr can +exercise, and PR 6 wires it. + +`attributes` is materialised once and does *not* track reads. Tracking is a +`CFVariable` concern, because `CFReader` builds a second `CFVariable` over +the same backing variable when it promotes one, and the two must keep +independent read sets (§4.3). `CFDatasetVariable.attributes` is therefore a +plain mapping, and `CFVariable.attributes` is a tracking view over it. + +`location` repeats `CFDataset.location` on the variable so that a variable +can name its own file without a back-reference to its dataset. It is +load-bearing: it is the `path` of a read proxy, and so part of the dask +array cache key, and it is what `LOAD_PROBLEMS.record()` reports. + +`closed` is on the interface because `Saver.complete()` has to refuse to run +until the file is released, and `isopen()` is netCDF vocabulary. A dataset +the caller opened is closed by the caller, so an implementation answers from +the store where it can, not only from its own flag. + +`create_dimension` accepts `size=None` to request an unlimited dimension; a +store with no such concept raises. `create_variable`'s `dimensions` defaults +to `()`, because grid-mapping variables are scalar. +``` + +- [ ] **Step 2: Note the `CFVariable` attribute view in spec §4.3** + +§4.3 says `CFVariable.__getattr__` resolves "against `self.attributes`" but +does not say what `self.attributes` *is*. Add, at the end of §4.3: + +```markdown +`CFVariable.attributes` is a `TrackedAttributes` view, constructed per +`CFVariable` over a snapshot of `cf_data.attributes`. It records reads — +which is how `cf_attrs_unused()` decides what reaches `cube.attributes` — +and it is a snapshot rather than a live view because a backend's attribute +mapping may write through to the file, and loading must never write. +``` + +- [ ] **Step 3: Record the deferred and the deliberate in §12.3 and §12.4** + +Add to §12.3, the open questions table: + +```markdown +| Q8 | `_add_grid_mapping_to_dataset` sets sixty-three CF grid-mapping parameters by Python attribute assignment, bypassing the ASCII-to-bytes coercion every other saved attribute goes through — so `crs_wkt` is written as `NC_STRING` while `grid_mapping_name` is `NC_CHAR`. Should they be regularised? | Not in this programme. PR 2 kept them on a named netCDF4 handle (`grid_variable`) rather than change what lands in the file. Regularising is a one-line-per-parameter change with a real CDL diff, and belongs in its own pull request | §4.5 | Nothing | +``` + +Add to §12.4, the decision log. The five bullets below were the ones known +when this plan was written; Tasks 6 and 10 to 12 added eight more, so the +entry as landed has **thirteen**, and §12.6's count in Step 4 moves with +it. The added eight are: the five declared properties on `CFVariable` +(T6-5), the `ncattrs`/`getncattr` leniency (T6-1), `AttributeError` → +`KeyError` on a missing CF attribute, the synthesised `bounds` link +reaching `IRIS_RAW` (T10-3), a borrowed bare dataset now wrapped in an +`EncodedDataset` (T11-6), the three members an emulator must now provide +(T12-4, T12-7), `cf_patch` keeping netCDF4 objects (T12-5), and the two +test modules added to `.hooks/check_netcdf4_imports.py`'s +`_PERMITTED_SUFFIXES`. + +```markdown +**2026-09-24 — during PR 2** + +- `CFDatasetVariable.attributes` does not track reads; `CFVariable` does. + Two `CFVariable`s can share one backing variable, and they must not share + a read set. +- `CFVariable.attributes` is a snapshot. A backend attribute mapping writes + through to the file, and a read-mode load must not write. +- The `spans` gap §5 called a latent bug is unreachable from Iris: + `_NCZARR_SCALAR_DIMENSION` only ever appears alone, so the `len == 1` + guard it lacks can never fire. PR 2 characterises the behaviour instead of + changing it, and the "one behaviour change" §5 allows is spent elsewhere. +- Saving a coordinate attribute whose name collides with a netCDF4 Python + member — `shape`, `size`, `dtype`, `name`, `dimensions`, `mask` — now + writes it, where the `hasattr()` "don't clobber" check used to drop it + silently. This is the saver-side half of the same defect the `__getattr__` + rewrite fixes on the load side. +- `finalise()` is on the interface but unwired until PR 6. +``` + +- [ ] **Step 4: Update §12.1 and §12.6** + +In §12.1, the pull request table, replace the PR 2 row (leaving the number +blank until Step 8 has it): + +```markdown +| 2 | `CFDataset`, and the CF variable classes rewritten against it | Drafted | — | +``` + +In §12.6, the document history table, append: + +```markdown +| 2026-09-24 | PR 2 built. §4.2 reconciled with the implemented interface: `location`, `__len__` and `ndim` on the variable, `closed` on the dataset, `attributes` no longer tracking, `create_dimension(size=None)` and `create_variable(dimensions=())`. §4.3 says what `CFVariable.attributes` is. Q8 opened on the grid-mapping assignments; thirteen decisions logged. | +``` + +- [ ] **Step 5: Commit the spec** + +```bash +pre-commit run --files docs/superpowers/specs/2026-09-21-zarr-io-design.md +git add docs/superpowers/specs/2026-09-21-zarr-io-design.md +git commit -m "$(cat <<'EOF' +Reconcile the Zarr I/O spec with the interface PR 2 built + +Six things the spec's §4.2 sketch did not settle, which the +implementation had to. None is a correction: §4.2 called its own listing +"deliberately small: only what the CF variable classes actually need", +and this is what they turned out to need. + +PRs 3 to 6 are written against §4.2, so it has to be the real interface. + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +``` + +- [ ] **Step 6: Diff the full suite against the baseline** + +The plan's central claim is that nothing changes except the documented +behaviours — ten of them, not two; see the Definition of Done. +Demonstrate it. + +```bash +pytest -n auto lib/iris/tests -q > ~/pr2-suite-final.txt 2>&1 +sed -i 's/\x1b\[[0-9;]*m//g' ~/pr2-suite-final.txt; tail -3 ~/pr2-suite-final.txt +diff <(grep -E "^(FAILED|ERROR)" ~/pr2-suite-baseline.txt | sed 's/ - .*//' | sort) \ + <(grep -E "^(FAILED|ERROR)" ~/pr2-suite-final.txt | sed 's/ - .*//' | sort) +grep -oE "[0-9]+ (passed|failed|skipped|xfailed|xpassed|error)" \ + ~/pr2-suite-baseline.txt ~/pr2-suite-final.txt | tail -20 +``` + +Expected: the `diff` is empty, and the counts differ only by the tests this +pull request adds. The baseline here is `11724 passed, 66 skipped, 0 failed, +0 errors` (§6.2), so the `diff` being empty means **both sides have no +`FAILED` or `ERROR` lines at all** — which also means an empty `diff` proves +nothing on its own if the colour codes are still in the file, since +`grep -E "^(FAILED|ERROR)"` would match nothing either way. That is what the +`sed` on line 2 is for. + +So the real check is the last command. **Read the skip count**, not only the +failures: a test that stopped running is neither a pass nor a failure, so a +`FAILED|ERROR` diff cannot see it. `66 skipped` must still read `66 +skipped` — a rise means coverage was lost, almost always a missing +`OVERRIDE_TEST_DATA_REPOSITORY`, which must be set for both runs or neither. + +`~/pr2-suite-baseline.txt` is the run recorded in section 6, on the baseline +commit, before Task 1. **It does exist** — Task 13 confirmed both baseline +files by md5 and compared against them, and the human partner has stated +the §6.2 baseline is not to be re-measured. Do not run the block below: a +re-measured baseline cannot detect the behaviour change it exists to +detect, and a stash-and-checkout dance across a thirteen-commit branch is +not a step to improvise. If a baseline file is ever genuinely missing, stop +and report it rather than regenerating it. Kept only as the record of how +it was made: + +```bash +git stash --include-untracked +git checkout 2cc0b01d4 +pytest -n auto lib/iris/tests -q > ~/pr2-suite-baseline.txt 2>&1 +sed -i 's/\x1b\[[0-9;]*m//g' ~/pr2-suite-baseline.txt +git checkout zarr-io-pr-2 +git stash pop +tail -1 ~/pr2-suite-baseline.txt # must be 0 failed, 0 errors; if not, see §6.1 +``` + +- [ ] **Step 7: Confirm the boundary held** + +All four checks as first written were wrong, and Task 13 replaced them. +Each correction is explained under the block. + +```bash +# 1. cf/dataset.py imports no backend, at module level or inside a function. +grep -nE "^\s*(from|import)\s" lib/iris/fileformats/cf/dataset.py + +# 2. ...and names no backend in executable code; comments and docstrings +# are prose about the implementations, and are not a breach. +python -c " +import re, tokenize +hits = [] +with tokenize.open('lib/iris/fileformats/cf/dataset.py') as f: + for tok in tokenize.generate_tokens(f.readline): + if tok.type in (tokenize.COMMENT, tokenize.STRING): + continue + if re.search(r'netcdf|zarr|h5|dask', tok.string, re.I): + hits.append((tok.start[0], tok.string)) +print(hits or 'clean') +" + +# 3. Every remaining netCDF name in the CF package is one of four known +# kinds. Classify each line; do not expect no output. +grep -rn "netcdf" lib/iris/fileformats/cf/*.py | grep -v "^lib/iris/fileformats/cf/_reader.py" + +# 4. saver.py assumes netCDF in exactly the places Task 12 named. +grep -n "_dataset\.dataset\|cf_var_grid\.\|grid_variable\." \ + lib/iris/fileformats/netcdf/saver.py | wc -l + +# 5. Nothing in Iris takes the deprecated netCDF4 fallback. +pytest -W "error:Reaching netCDF variable member" \ + lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/netcdf/ -q +``` + +Expected: five import lines from the first — `abc`, `collections.abc`, +`types`, `typing`, `numpy`, and nothing else; exactly one hit from the +second, `deprecated_netcdf_member`, which spec §4.2 mandates; a +classifiable line for every hit of the third; a count of **68** from the +fourth — three `_dataset.dataset`, two `cf_var_grid.` and sixty-three +`grid_variable.`; and a clean pass from the fifth. + +Why each changed: + +- **Checks 1 and 2 replace `! grep -niE "netcdf|zarr|h5|dask" …`, which can + never print `clean`.** `cf/dataset.py` has matched on a dozen lines since + Task 1 built it, and every one is correct: the module docstring explains + what the netCDF and Zarr implementations do with `**encoding`, comments + name the concrete case a member exists for, and `deprecated_netcdf_member` + is a mandated ABC member. The claim worth checking is about imports and + executable code, not about a word appearing in prose. +- **Check 3 was "expect no output", which it can never give either.** Every + line is one of four kinds: (A) `dataset.py`'s own prose and its mandated + member; (B) the `_deprecated_netcdf_member` reach-through in + `_variables.py`, its call site and its docstrings; (C) docstring + cross-references to `netcdf._dataset.NetCDFDatasetVariable` and + `loader._add_unused_attributes`; (D) prose this PR never touched, in + `_variables.py` and `_group.py`. A bare "expect no output" cannot tell a + site that was converted from a site that was missed, so make it an + allow-list: classify every line, and treat a line fitting none of the four + as the finding. The *count* is not a finding — Tasks 11 and 12 + legitimately add and remove docstring references. +- **Check 4's expected count was 66.** The arithmetic rested on sixty-one + grid-mapping assignments; there are sixty-three (Ruling T12-3), so the + total is 68. +- **Check 5 replaces `-W error::iris._deprecation.IrisDeprecation`, which + cannot run at all.** It exits 4 during conftest import, because + `lib/iris/tests/graphics/__init__.py` calls `warn_deprecated` for + `GraphicsTestMixin` at import time. That file and `lib/iris/_deprecation.py` + are byte-identical to the merge base, so it is pre-existing and unrelated + to this PR — and it means the check as written was never runnable. The + message-targeted filter above is narrower *and* stronger: it turns only + this PR's own reach-through warning into an error, so an unrelated + pre-existing deprecation cannot mask it. Ruling T11-9 declined to add the + filter to `pyproject.toml`, so it is a local check, not a CI gate. + +- [ ] **Step 8: Open the pull request** + +```bash +git push -u origin zarr-io-pr-2 +gh pr create \ + --repo SciTools/iris \ + --base brownfield \ + --head trexfeathers:zarr-io-pr-2 \ + --label "Agentic" \ + --label "Type: Feature Branch" \ + --title "Zarr I/O PR 2: CFDataset, and the CF variable classes rewritten against it" \ + --body "$(cat <<'EOF' +Part of #6977. Second of seven pull requests into `brownfield`; see +`docs/superpowers/specs/2026-09-21-zarr-io-design.md` §5 for the programme +and §12.1 for its state. The implementation plan is in +`docs/superpowers/plans/2026-09-24-zarr-io-pr2.md`, and ships with it. + +## What this does + +Introduces `iris.fileformats.cf.dataset`: a pair of narrow abstract classes, +`CFDataset` and `CFDatasetVariable`, describing what Iris needs from an array +store. `iris.fileformats.netcdf._dataset` implements them over netCDF4, and +everything that used to reach through `CFVariable` into a `netCDF4.Variable` +now goes through the interface instead — the CF reader, the load rules, the +netCDF loader, and the netCDF saver. + +The change that will be most visible to a reader is `CFVariable.__getattr__`. +It resolved an attribute name against the file *and* against the backing +netCDF4 object, and cached the answer with `setattr`. It now resolves against +`self.attributes` only. A CF attribute called `shape`, `dtype` or `name` used +to be shadowed by the netCDF4 member of that name and silently lost; it +reaches the cube now. + +## Behaviour changes + +Ten, all deliberate. Items 1 to 7 and items 9 and 10 change what Iris +**does**. Item 8 changes what Iris **requires of its inputs**, which is still +a behaviour change from the caller's side — a save that used to work now +raises — so it is in the list rather than in a compatibility footnote. + +This list has been declared complete three times and grown three times — +two, then eight, then nine, now ten — each time because a review enumerated +something the previous pass had walked past, not because the code changed +underneath it; read "ten" as the number three reviews have reached, not as a +number anyone has proved. + +Each item names the test that pins it. Every one of those tests was checked +by reverting the behaviour in the source and confirming the test fails; where +that check found no coverage, the item says so plainly instead of implying +there is a pin. + +1. **A CF attribute whose name collides with a member name is no longer + lost.** One headline, two unrelated mechanisms, so they are listed + separately. + + - *On save* (finding F13). The "don't clobber" guard in + `Saver._set_cf_var_attributes` was `hasattr(cf_var, name)` and is now + `name not in cf_var.attributes`. An attribute named `shape`, `size`, + `name`, `dtype`, `dimensions` or `mask` is written to the file, where the + guard used to see the netCDF4 member of that name and drop the attribute + silently. + *Pinned by* `lib/iris/tests/integration/netcdf/test_attributes.py::TestAttributesNamedLikeNetcdf4Members::test_attributes_named_after_netcdf4_members_are_saved`. + Restoring the `hasattr` form fails it. + - *On load*, for names that are **not** declared `CFVariable` properties. + `__getattr__` resolves against `self.attributes` instead of falling + through to the netCDF4 object, so the file's value wins and is recorded + as used rather than being consumed and discarded. + *Pinned by* `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py::TestShadowedAttributeNames`, + parametrised over `SHADOWED_NAMES = ["filename", "cf_name", "spans", + "attributes", "cf_data"]`. Filtering those names out of the attribute + mapping fails 15 of the class's 17 tests. + +2. **The five declared-property names behave differently again — a different + mechanism with the same headline.** `dimensions`, `shape`, `ndim`, `dtype` + and `size` are declared properties on `CFVariable` and still resolve to the + *storage* object: `cf_var.shape` returns the array shape, not a file + attribute called `shape`. What changed is that the read is no longer + *recorded*. The old `__getattr__` marked such a name as read before + forwarding to the netCDF4 variable, so `_add_unused_attributes` dropped it + and a legitimate user attribute vanished from the loaded cube — even though + the value returned was never the file's. A declared property short-circuits + `__getattr__` entirely, so nothing is recorded and the attribute survives + onto the cube. This is emphatically *not* "the shadowing was removed": the + member still wins. + *Pinned by* `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py::TestTypedProperties::test_a_file_attribute_colliding_with_a_typed_property`, + parametrised over all five. Making each property record its own name fails + all five parametrisations. + +3. **`CFVariable.attributes` is a snapshot taken at construction**, so nothing + on the load path can write to the file it is reading. + *Pinned by* `lib/iris/tests/unit/fileformats/cf/test_CFVariable.py::TestAttributeAccess::test_attributes_are_a_snapshot_taken_at_construction`, + with `lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py::TestMutation` + underneath it for the write-through semantics that make the snapshot + necessary. Wrapping the backend mapping live instead of copying it fails + the first; making `TrackedAttributes.__setitem__` not write through fails + the second. + +4. **A malformed file that used to raise now loads.** Where `ncattrs()` lists + a name that `getncattr()` then refuses — which the netCDF4 library does for + some malformed files — `_NetCDFAttributes` substitutes `""` and the load + succeeds. `cf/_reader.py`'s old `_getncattr` already tolerated it with the + same default; the leniency now lives one layer down, so the same malformed + file gets one answer rather than two different ones depending on which + layer read it. + *Pinned by* `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py::TestAttributes::test_unreadable_attribute_becomes_empty_string`. + Removing the `except AttributeError` fails it. + +5. **The exception type on a missing CF attribute moved `AttributeError` → + `KeyError`.** Call sites that read a CF attribute by attribute access + (`cf_var.long_name`) now subscript the mapping + (`cf_var.attributes["long_name"]`), and a miss raises `KeyError` where it + used to raise `AttributeError`. Minor, but downstream code catching + `AttributeError` around a CF read will see a different exception. There + are six such sites — four on the load path, and two on the **save** path, + which is worth stating separately because a reader of this list would not + think to look there: + + - *Load.* `cf/_reader.py:493-494` reads `standard_name` and then + `long_name`; with `FUTURE.derived_bounds` off — the default — **both** + subscripts are unguarded, and with it on, `standard_name` is guarded by + an `in` test three lines above while `long_name` is not. + `_nc_load_rules/helpers.py:1158` reads `grid_mapping_name`. + `netcdf/ugrid_load.py:380` reads `node_coordinates`, and `:406` reads + the connectivity's `cf_role`, inside an `assert`. + - *Save.* `netcdf/saver.py:1159` reads `standard_name` and `:1187` reads + `coordinates`, both while rewriting a dimensionless vertical + coordinate's `formula_terms`. + + A seventh converted read, `_nc_load_rules/helpers.py:567`, is deliberately + not in that list: its `contextlib.suppress` was changed from + `AttributeError` to `KeyError` along with the read, which is exactly the + handling the six above did not get. + + *Pinned by* `lib/iris/tests/unit/fileformats/cf/test_CFReader.py::Test_translate__formula_terms_derived_bounds::test_raises_when_standard_name_empty_and_long_name_absent` + — **for the `cf/_reader.py` site only**, and there for its `long_name` + subscript. The other five have no test that fails if reverted. That is + measured, not assumed, and measured site by site rather than extrapolated: + softening each subscript to `.get()` and running the whole of + `lib/iris/tests/unit/fileformats/` and `lib/iris/tests/integration/` left + the suite green every time — same collected total as the unmutated + control, no new failure at any of the five. + + All five are malformed-input paths — a UGRID mesh variable with no + `node_coordinates` or no connectivity attribute, a grid-mapping variable + with no `grid_mapping_name`, a `formula_terms` variable with no + `standard_name`, a cube variable with no `coordinates` — and none is + reachable from a well-formed file. **They ship flagged rather than + fixed.** Closing them means writing new tests for five malformed-input + paths into the final commit of a branch whose whole claim is that it + changed nothing; disclosing them here is the better trade. A follow-up + issue covering all six sites — settling once whether a missing CF + attribute should raise, and which exception — would sit naturally with the + merge-back changelog; this pull request proposes one rather than opening + it. + +6. **`IRIS_RAW` differs on the error-recovery path.** `CFReader` now writes + its synthesised and invalidated `bounds` links into the attributes mapping + rather than onto the `CFVariable` object, and `build_raw_cube` reads that + mapping unfiltered. A formula-term or derived-bounds variable that falls + back to `build_raw_cube` gains a `bounds` key it did not have — and where + the file carried its own `bounds`, the invalidation writes `None` over it, + so `IRIS_RAW` shows `None` where it used to show the file's value. Narrow, + but real. + + *The write is pinned; the behaviour change is not.* Deleting the + invalidating write fails + `lib/iris/tests/integration/netcdf/derived_bounds/test_bounds_files.py::test_load_legacy_hh[with_db]` + and `lib/iris/tests/unit/fileformats/cf/test_CFReader.py::Test_translate__formula_terms_derived_bounds::test_promotes_non_formula_root_bounds_to_data`. + But replacing it with the pre-PR form — `root_var.bounds = None`, a plain + Python attribute that never reaches the mapping and so never reaches + `IRIS_RAW` — leaves **every** test green, including + `test_CFReader.py::TestSynthesisedBoundsLink`, whose own docstring is + explicit that it characterises `TrackedAttributes` rather than exercising + `_reader.py`. So this item joins item 5 as declared-but-unpinned, and for + the same reason it is flagged rather than fixed here. + +7. **A borrowed *bare* `netCDF4.Dataset` now decodes character data.** + `CFReader` and `iris.load` both accept an already-open dataset. Where that + object is **not** already one of Iris's wrappers — what the ncdata / + Xarray bridge passes — `NetCDFDataset.from_existing` now wraps it in an + `EncodedDataset`, where `CFReader` previously stored it untouched. Checked + against a real pre-branch checkout: `cf_group["labels"][:]` goes + `|S1 (3, 4)` → ` Saving through a ``CFDataset`` asks an emulating object for three members + > that the saver never used to reach for. All three are public netCDF4 API, + > and all three are now required of an emulator: + > + > * ``Dimension.size`` - the CF layer reads dimension lengths through it. + > ``__len__`` is not an alternative, as ``_EmulatedDimension`` explains. + > * ``Dataset.ncattrs()`` - read once, as the dataset is wrapped. + > * ``Variable.ncattrs()`` - read once per variable, as each is wrapped. + > + > They are marked below where the emulators define them. This is an + > accepted consequence of the change, not an oversight: there is + > deliberately no fallback for an emulator that lacks them. + + And, at the `_EmulatedDimension` definition site, on why `len()` will not + serve: + + > netCDF4.Dimension.size, which is how the length is read back. len() is + > not an alternative: the emulator is put inside a _thread_safe_nc wrapper, + > whose __getattr__ forwards named members but is never consulted for the + > len() protocol. + + The absence of a shim is deliberate: a speculative + `getattr(dim, "size", None)` fallback was declined as defensive wrapping + for an unconfirmed problem, which `lib/iris/AGENTS.md` bans. + *Pinned by* the emulator doubles and the saves through them in + `lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py`. + Deleting any one of the three members makes all five emulator saves error + — confirmed for each of the three separately. This is the only one of the + ten whose pin is a **test double** rather than a behavioural assertion, + which is appropriate: the thing being pinned is an interface demand on + third-party code, and the double is Iris's only standing description of + what that code must provide. It is *not* a live ncdata test. + + **A question for a maintainer before this lands:** none of this is + verified against real ncdata. ncdata is not installed in the development + environment used here, and installing it needs your say-so. Is an ncdata + round-trip check wanted first? If ncdata turns out to lack any of the + three members, the fix is one `getattr` in the dataset layer, not a + redesign. + +9. **Reaching a netCDF4 member through a `CFVariable` now warns.** The old + `__getattr__` answered any name the backing netCDF4 object carried — + `value = getattr(self.cf_data, name)`, silently. The route survives for + one deprecation cycle, but it now goes through + `NetCDFDatasetVariable.deprecated_netcdf_member`, which issues an + `IrisDeprecation` naming the member and pointing at + `cf_var.cf_data.variable`. `CFVariable` is exported in + `iris.fileformats.cf.__all__`, so this reaches callers outside Iris: + third-party code that read `cf_var.getncattr` or `cf_var.group` gets a + warning where it got silence, and under `-W error` a read that worked + before now raises. A `hasattr()` probe that comes back **True** warns too, + because the member was fetched; one that comes back False does not, + because nothing was used. Nothing in Iris itself takes the route — see + the reach-through run below, which measures that at zero. + *Pinned by* `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py::TestDeprecatedNetcdfMember::test_a_reach_through_warns` + and `…::TestDeprecatedNetcdfMember::test_the_message_names_the_replacement`, + both `pytest.warns(IrisDeprecation)`, with + `…::TestDeprecatedNetcdfMember::test_a_missing_name_raises_and_does_not_warn` + holding the probe boundary. Deleting the `warn_deprecated` call fails the + first two and leaves the third passing. + +10. **A dataset the caller hands to `iris.save` is now mutated.** + `NetCDFDataset.from_existing` calls `set_auto_chartostring(False)` on the + object it borrows, so saving into a caller's already-open dataset turns + auto-chartostring off on it, permanently, as a side effect of the save. + The merge base's **save** path never called it. Two qualifications, both + narrowing: the **load** path already did exactly this before this branch + (`cf/_reader.py` applied the same call to datasets it borrowed), so the + change is to the save path alone; and it is inert for anything + `from_existing` has to wrap — a bare `netCDF4.Dataset`, or an emulator — + because the wrapping `EncodedDataset` does its own decoding and swallows + the call rather than forwarding it. What is left is a save into an + object that is already an Iris thread-safe wrapper, where the call + reaches the real netCDF4 dataset underneath. Iris's saver reads no + character data, so nothing here does anything with the setting; it is + listed because a permanent change to an object the caller still owns is + caller-visible whether or not it matters. + *Pinned by* `lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py::TestAutoChartostring::test_turned_off_on_a_borrowed_dataset`, + which spies on the class rather than the instance because + `DatasetWrapper.__setattr__` forwards instance-level patching to the + contained object. Removing the call fails it. + +An eleventh that is *not* a change: spec §5 allowed one behaviour fix, to +`CFVariable.spans`. Reading the code showed the gap is unreachable — +`_NCZARR_SCALAR_DIMENSION` only ever appears as a variable's sole dimension — +so this pull request characterises the current behaviour rather than altering +it, and spends the allowance on (1) instead. + +## Five things a reviewer will want stated + +**`cf_patch` still receives netCDF4 objects.** The public +`iris.site_configuration["cf_patch"]` hook has been documented since Iris 1.3 +as being handed netCDF4 objects. Both arguments — the dataset and the +variable — are unwrapped before the call, so a `CFVariable` never reaches it. +An interim version of this branch passed a `NetCDFDatasetVariable`, which +defines no `__setattr__` and so would have swallowed `variable.name = value` +silently; that was caught and fixed, and is now pinned by +`lib/iris/tests/integration/netcdf/test_attributes.py::TestCfPatch::test_cf_patch_receives_a_netcdf4_variable`. + +**`.hooks/check_netcdf4_imports.py` gained two entries.** The repository +forbids `import netCDF4` outside `_thread_safe_nc`; +`lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py` and +`lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py` are added to +its `_PERMITTED_SUFFIXES` because they must construct a real, bare +`netCDF4.Dataset` to test the wrapping path that exists for exactly such an +object. This widens a lint allow-list for two test modules; it is not a +behaviour change. + +**Where a file attribute and an undeclared netCDF4 member share a name, the +file wins now.** That is the load half of behaviour change 1, stated the +other way round, and it is worth saying explicitly because it inverts a +precedence. The stronger statement is that nothing inside Iris depends on +that precedence any more, in either direction: after this branch no reader in +the library reaches an undeclared netCDF4 member through a `CFVariable` at +all. That is measured rather than asserted — the reach-through run below +promotes this branch's own deprecation warning to an error and gets zero +occurrences. (`ugrid_load.py`'s two error messages did read `cf_var.name`, +which would have been exactly such a reach; this branch moved them to +`cf_var.cf_name`, Iris's own name for the variable.) So the inversion is +visible only to a caller outside Iris, reading a colliding name from a +`CFVariable` — and only for a file perverse enough to carry a CF attribute +named after a netCDF4 member. + +**Six assertions in +`lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py` +changed, and that is not a loosening.** They were only ever reachable through +`MagicMock` auto-vivification — the double returned a `MagicMock` for any +attribute, so the assertions described the mock, not a variable. Re-measured +against a real backing variable across 36 cases, the old and new forms +diverge in none of them; the new form describes what a real `CFVariable` +does. + +**Eight grid-mapping parameter assignments gained coverage they never had** — +Mercator's `scale_factor_at_projection_origin` and the seven in the +PolarStereographic branch, in +`lib/iris/tests/integration/netcdf/test_attributes.py::TestGridMappingAttributes`. +These are *tests added*, not behaviour changed: a round-trip against the +pre-branch tree confirmed the values were always reaching the file. + +## How the negative claim was checked + +- **Full suite against the merge base.** Baseline (recorded on `2cc0b01d4` + before any of this work): `11724 passed, 66 skipped, 6482 warnings`, with + zero `FAILED` and zero `ERROR` lines. This branch: `11992 passed, 66 + skipped, 6483 warnings`, also zero and zero. The normalised diff of the + `FAILED`/`ERROR` lines is empty on both sides. **Read nothing into the + warning totals.** Five runs of one unchanging tree gave 6483, 6485, 6484, + 6483, 6483 — the summary names the same 13 warning sites every time, but + import-time module deprecations are counted once per `-n auto` worker that + imports the module, so the total moves with how the tests happened to be + distributed. An earlier draft of this section explained the 6482 → 6483 + step as one extra `iris.save`; that was a cause invented for a difference + inside run-to-run noise, and it is withdrawn. The pass and skip counts are + stable across those runs and are the ones carrying the claim. The skip + count is read deliberately and is unchanged — a test that silently stopped + running is neither a pass nor a failure, so a failure diff cannot see it. +- **+268 passed, accounted for exactly.** 200 tests in new modules, plus 67 + added in modified modules (counting parametrisations), plus 6 from two + inherited mixin methods reaching three modules, minus 5 removed. No + `parametrize` decorator was removed and no existing parametrisation was + widened, so the arithmetic is not hiding a substitution. +- **The boundary.** `cf/dataset.py` imports only `abc`, `collections.abc`, + `types`, `typing` and `numpy`; a token-level scan excluding comments and + docstrings finds exactly one backend name in executable code, + `deprecated_netcdf_member`, which spec §4.2 mandates. `saver.py` contains + no `import netCDF4` and no `_setncattr`, and names netCDF4 in exactly 68 + places: three `self._dataset.dataset` escape hatches (two `file_format` + reads and the `cf_patch` hand-off), two `cf_var_grid.` and sixty-three + `grid_variable.` grid-mapping assignments. +- **The deprecating reach-through fires nowhere in Iris's own paths.** + Running `lib/iris/tests/unit/fileformats/` and + `lib/iris/tests/integration/netcdf/` with the reach-through warning + promoted to an error produces zero occurrences. + +## Not here + +No changelog fragment: feature-branch pull requests defer those to the +merge-back, along with the closing keywords. See §5 of the spec. Behaviour +change 9 — the deprecation on the netCDF4 reach-through — is the kind of +change a fragment would normally accompany, and a follow-up issue for the +six `KeyError` sites in behaviour change 5 is proposed above; both belong +with the merge-back, where the fragment is written. + +The sixty-three grid-mapping parameter assignments in `saver.py` keep a +netCDF4 handle rather than move to `.attributes`, because they bypass the +ASCII-to-bytes coercion and moving them would change the CDL — `crs_wkt` +would stop being `NC_STRING`. Recorded as Q8 in the spec. + +One unrelated defect was found and is **not** fixed here, and is not a flake. +Running `lib/iris/tests/unit/fileformats/netcdf/loader/test_load_cubes.py` +before `lib/iris/tests/integration/netcdf/test_coord_systems.py` in the same +process errors all ten tests in the latter. `test_coord_systems.py` imports +the unit module as `tlc`, and its module-scoped autouse fixture only reaches +its `yield` inside `if not hasattr(tlc, "TMP_DIR")`, while the unit module +sets that global at import. Unit module first ⇒ the fixture returns without +yielding ⇒ every test in the module errors. It is deterministic on module +ordering, not on scheduling. All three files are byte-identical to the merge +base, so it predates this branch — but it is a latent CI failure under an +unlucky ordering and is worth an issue of its own. + +🤖 Generated with [Claude Code](https://claude.com/claude-code) +EOF +)" +``` + +- [ ] **Step 9: Record the pull request number** + +Put the number and link into §12.1 of the spec, and push: + +```bash +git add docs/superpowers/specs/2026-09-21-zarr-io-design.md +git commit -m "$(cat <<'EOF' +Record PR 2 in the Zarr I/O progress table + +Part of #6977 + +Co-Authored-By: Claude Opus 5 +EOF +)" +git push +``` + +--- + +## 6. Verification + +The pull request's central claim is a *negative* one: nothing changes except +the two documented behaviours. A negative claim cannot be shown by a green +run — only by a run that matches a run taken before any of this existed. So +the baseline comes first, before Task 1, and "green" below always means +**matches the baseline**. + +On this machine, after F15 was fixed, the baseline happens to *be* green: +`11724 passed, 66 skipped, 0 failed, 0 errors`. That is a gift, not a +licence to stop comparing. Two things still separate "green" from +"unchanged", and both are invisible to a failure count: + +- **The skip count.** A test that stops being collected takes its coverage + with it and leaves no failure behind. 66 is the number; it must not rise. +- **The passed count.** It should rise by exactly the number of tests this + pull request adds, and §6.4 makes you enumerate them. + +So read the whole tail line, not the colour. + +### 6.1 Before Task 1: prove the environment, then record the baseline + +```bash +source "$(conda info --base)/etc/profile.d/conda.sh" +conda activate iris-dev +echo "${OVERRIDE_TEST_DATA_REPOSITORY:?set OVERRIDE_TEST_DATA_REPOSITORY first}" +ls "$OVERRIDE_TEST_DATA_REPOSITORY/NetCDF/unstructured_grid/" | wc -l +``` + +That last line is not ceremony. **It must print 13. If it prints a number +under 10, stop and update `iris-test-data` before going any further** — + +```bash +git -C ../iris-test-data fetch upstream # add the remote first if absent: + # https://github.com/SciTools/iris-test-data.git +git -C ../iris-test-data merge --ff-only upstream/master +``` + +— because the UGRID files live there, and +`lib/iris/tests/integration/netcdf/test_ugrid_load.py` is the only end-to-end +coverage Task 9's `ugrid_load.py` rewrite has. Without them those thirteen +tests fail on arrival, look like pre-existing noise, and go on failing for +the same reason after the rewrite has broken something real. That is exactly +the trap `lib/iris/tests/AGENTS.md` warns about, wearing a failure's clothes +instead of a skip's. It had already sprung on this machine when the plan was +written; see finding F15. + +Then, on the baseline commit and with the working tree clean: + +```bash +git status --short # must be empty; the plan file itself is untracked +git rev-parse --short HEAD # must be 2cc0b01d4 +pytest -n auto lib/iris/tests -q > ~/pr2-suite-baseline.txt 2>&1 +sed -i 's/\x1b\[[0-9;]*m//g' ~/pr2-suite-baseline.txt +tail -1 ~/pr2-suite-baseline.txt +``` + +`sed` strips the colour codes, without which `grep -E "^(FAILED|ERROR)"` +matches nothing and every later diff comes back empty and meaningless +**[verified: this caught a false clean while writing this plan]**. + +Record the last line somewhere you will still have it in a day. Then do the +same for the tier most tasks below use, which is not a subset of the same +numbers: + +```bash +pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/ -q \ + > ~/pr2-tier-baseline.txt 2>&1 +sed -i 's/\x1b\[[0-9;]*m//g' ~/pr2-tier-baseline.txt +``` + +### 6.2 What the baseline was on the day this plan was written + +Measured here, on `zarr-io-pr-2` at `2cc0b01d4`, four cores, `-n auto`, with +`iris-test-data` at `f4a6a05` **[verified]**: + +| Run | Result | +|---|---| +| `lib/iris/tests` | `11724 passed, 66 skipped, 6482 warnings in 247.29s` — **0 failed, 0 errors** | +| `unit/fileformats/` + `integration/` | `4465 passed, 26 skipped, 2425 warnings, 10 errors in 160.05s` | + +And the smaller selections the tasks below actually run, all serial, all +**[verified]**: + +| Selection | Baseline | +|---|---| +| `unit/fileformats/cf/` | `306 passed, 1 skipped` | +| `unit/fileformats/netcdf/` | `374 passed, 1 skipped` | +| `unit/fileformats/nc_load_rules/` | `332 passed, 0 skipped` | +| `unit/mesh/` + `integration/mesh/` | `779 passed, 0 skipped` | +| `integration/netcdf/` | `206 passed, 21 skipped` | +| `integration/test_cf.py` | `18 passed, 0 skipped` | + +Three things to know before you read your own numbers against these. + +**The one skip in `unit/fileformats/cf/` and the one in +`unit/fileformats/netcdf/` are expected, and they are not the same kind.** +The first is `test_CFReader.py`, *"Test only applicable when FUTURE context +is enabled"* — unconditional in this configuration, so it will be there for +you too. The second is `attribute_handlers/test_GribParamHandler.py`, +*"could not import 'iris_grib'"* — environment-dependent, so if you have +`iris_grib` installed you will see `375 passed, 0 skipped` instead. Either +is fine. **A second skip appearing in either directory is not**, and is +almost always a rename that left a stale import or a module now raising at +import time and being swallowed. + +**The ten errors in the tier run are finding F14** — all in +`integration/netcdf/test_coord_systems.py`, all +`ValueError: _setup did not yield a value`, a pre-existing fixture defect +whose firing depends on whether +`unit/fileformats/netcdf/loader/test_load_cubes.py` was imported into the +same process first. (Corrected in Task 13: that is module *import order*, not +xdist scheduling — it reproduces serially with no xdist at all. Under +`-n auto` it is which modules share a worker that decides the order.) They do +not appear in the full-suite run, and they do not appear when +`integration/netcdf/` is run serially. They are not yours and chasing them +will cost you an afternoon. Confirm they are in your baseline too, and move +on. + +**Everything else is green, and that is recent.** Before `iris-test-data` +was brought up to date this table read `27 failed … 11 errors`, all missing +data files (F15). If your full-suite run is not at `0 failed, 0 errors`, +your test data is stale — go back to §6.1 rather than treating the red as +background. + +### 6.3 During the work: what each task compares against + +Every task's own steps name the tier to run; §5.0 is the ritual around it. +Two rules govern reading the result. + +**Unit directories must be literally clean — 0 failed, and the skip count +§6.2 records for that exact selection.** `lib/iris/tests/unit/` needs no +external data, so there is nothing to excuse a failure. The skip count is +not always zero, though: `unit/fileformats/cf/` carries 1 and +`unit/fileformats/netcdf/` carries 1, both named in §6.2, and a step below +that says "0 failed, 1 skipped" means that pre-existing one and no other. +A *new* skip in a unit directory is a test that stopped being +collected — almost always a rename that left a stale import, or a module +that now raises at import time and is being swallowed. Treat it as a +failure, because that is what it is. + +**Anything touching `lib/iris/tests/integration/` is compared, not judged.** + +```bash +pytest -n auto lib/iris/tests/unit/fileformats/ lib/iris/tests/integration/ -q \ + > ~/pr2-tier-now.txt 2>&1 +sed -i 's/\x1b\[[0-9;]*m//g' ~/pr2-tier-now.txt +diff <(grep -E "^(FAILED|ERROR)" ~/pr2-tier-baseline.txt | sed 's/ - .*//' | sort) \ + <(grep -E "^(FAILED|ERROR)" ~/pr2-tier-now.txt | sed 's/ - .*//' | sort) +``` + +Empty diff, and a skip count no higher than 6.2's, is the pass condition. +Sorting and dropping the message keeps xdist's nondeterministic ordering and +varying tracebacks out of the comparison. + +### 6.4 At the end: the whole-suite diff + +Task 13 Step 6 runs it, and Task 13 Step 7 checks the architectural boundary +held. The pass condition is the same shape as 6.3's, over +`lib/iris/tests`, plus: the counts differ only by the tests this pull request +adds. Those are enumerable, so enumerate them — + +```bash +git diff --stat 2cc0b01d4 -- "lib/iris/tests/**/test_*.py" +``` + +— and the passed count should rise from 11724 by the number of test +functions added, no more and no less, with `66 skipped` and `0 failed, +0 errors` unchanged. A larger rise means a parametrisation was widened +somewhere; a smaller one means something stopped being collected. + +Note `git diff --stat` counts *files*, not functions, so read the diff +itself for the new `def test_` lines: + +```bash +git diff 2cc0b01d4 -- "lib/iris/tests/**" | grep -cE "^\+\s*def test_" +``` + +### 6.5 The four things a full-suite diff still cannot see + +Each has its own check, listed here so none is assumed to be covered by the +big run: + +1. **An attribute that now reaches a cube, or stops reaching one.** Nothing + fails; the cube is simply different. Covered by + `integration/test_netcdf__loadsaveattrs.py` and `integration/test_cf.py`, + which Tasks 6 and 10 run explicitly, and by Review Focus 1 and 3. +2. **A CDL change in a saved file.** Covered by the `assert_CDL` comparisons + in `test_netcdf.py` and `integration/netcdf/`, which Task 12 runs, and + deliberately sidestepped for the grid-mapping parameters (F8). +3. **A deprecation warning Iris trips on its own paths.** Covered by the + `-W "error:Reaching netCDF variable member"` run in Task 13 Step 7, + which replaced `-W error::iris._deprecation.IrisDeprecation`: that form + exits 4 during conftest import on a pre-existing, unrelated deprecation, + so it never ran. See Step 7 for the detail. +4. **The boundary itself** — a netCDF name leaking into `cf/dataset.py`, or a + backend reach-through appearing somewhere other than the three named + escape hatches. Covered by Task 13 Step 7's greps, which are exact counts, + not spot checks. + +--- + +## 7. Definition of done + +- [ ] All thirteen tasks complete, each committed separately, no commit made + with `--no-verify`. +- [ ] `lib/iris/fileformats/cf/dataset.py` exists, imports nothing but the + standard library and numpy, names no backend in *executable* code + except the ABC's own mandated `deprecated_netcdf_member`, and its two + ABCs match spec §4.2 exactly. (Corrected in Task 13. The original + bullet said the file "contains no reference to netCDF, Zarr, HDF5 or + Dask", which has been false since Task 1 built the file this plan + specified: the module docstring explains what the netCDF and Zarr + implementations do with `**encoding`, several comments name the + concrete case a member exists for, and `deprecated_netcdf_member` is + required by §4.2. A word-level grep cannot distinguish prose from a + boundary breach; the import list and the executable-token check can.) +- [ ] `lib/iris/fileformats/netcdf/_dataset.py` implements both, and the + shared contract test from Task 5 runs against it. +- [ ] `CFVariable.__getattr__` resolves against `self.attributes` only, with + no `setattr` caching, and a miss raises `AttributeError` after a single + deprecating reach-through. +- [ ] `grep -rn "getattr(cf_\|hasattr(cf_" lib/iris/fileformats/cf/ + lib/iris/fileformats/_nc_load_rules/ lib/iris/fileformats/netcdf/` + returns only the sites §1 lists as out of scope. +- [ ] `saver.py` contains no `import netCDF4`, no `_setncattr`, and exactly + three `self._dataset.dataset` escape hatches plus the grid-mapping + handle. +- [ ] The full suite is `0 failed, 0 errors` — as it is at baseline — and + still reads `66 skipped`, with the passed count up by exactly the + number of tests this pull request adds (§6.4). +- [ ] The **ten** deliberate behaviour changes are each pinned by a test + that fails if reverted. (Corrected in Task 13, counted from the + ledger's rulings rather than from this plan, which has been wrong + about the number twice: it said two, then seven. The ten are + member-name collisions on load and on save (F13), the five declared + properties no longer recording a read (T6-5), the `attributes` + snapshot, `""` for an unfetchable attribute (T6-1), `AttributeError` + → `KeyError` on a missing CF attribute, the synthesised `bounds` link + reaching `IRIS_RAW` (T10-3), a borrowed *bare* `netCDF4.Dataset` now + decoding character data (T11-6), the three members a netCDF4 + *emulator* must now provide (T12-4, T12-7), and — added in Task 13's + first fix round, an omission from the enumeration rather than a + re-opening of the count — the netCDF4 reach-through through a + `CFVariable` now emitting an `IrisDeprecation` where it used to be + silent, which reaches external callers because `CFVariable` is in + `iris.fileformats.cf.__all__`; and — added in the final review's fix + round, again an omission from the enumeration rather than a change to + the code — `NetCDFDataset.from_existing` turning auto-chartostring off + on a dataset the caller handed to `iris.save`, which the merge base's + save path never did.) +- [ ] Two of the ten are only partly pinned, and the pull request body + says so rather than implying coverage: `AttributeError` → `KeyError` + is pinned at the `cf/_reader.py` site and at none of the other five + (corrected in Task 13's first fix round, which found six sites, not + three — two of them on the *save* path, in `saver.py` — and measured + each of the three new ones separately); and T10-3's *consequence* for + `IRIS_RAW` is untested, though the write itself is. All were measured + by reversion, not assumed. +- [ ] All five Review Focus items have a passing test in the task that owns + the code. +- [ ] Spec §4.2, §4.3, §12.1, §12.3, §12.4 and §12.6 updated (Task 13). +- [ ] No changelog fragment, and the pull request body says so. +- [ ] Pull request open against `SciTools/iris:brownfield`, labelled `Agentic` + and `Type: Feature Branch`, body containing "Part of #6977" and no + closing keyword. +- [ ] This plan is on the branch, with any step the implementation proved + wrong corrected in place. diff --git a/docs/superpowers/specs/2026-09-21-zarr-io-design.md b/docs/superpowers/specs/2026-09-21-zarr-io-design.md index 3ca78dd44f..cecc2d46f4 100644 --- a/docs/superpowers/specs/2026-09-21-zarr-io-design.md +++ b/docs/superpowers/specs/2026-09-21-zarr-io-design.md @@ -7,7 +7,7 @@ | | | |---|---| | **Phase** | Design, awaiting approval | -| **Progress** | 1 of 7 pull requests raised ([#7298](https://github.com/SciTools/iris/pull/7298), in review); merge-back not started — see §12.1 | +| **Progress** | 2 of 7 pull requests raised ([#7298](https://github.com/SciTools/iris/pull/7298) and [#7303](https://github.com/SciTools/iris/pull/7303), both in review); merge-back not started — see §12.1 | | **Next action** | Spec approval, then the implementation plan | | **Blocked on** | Nothing | | **Branch** | `zarr-io-design` on `bjlittle/iris`, targeting `SciTools/iris:brownfield` | @@ -274,34 +274,42 @@ need. class CFDatasetVariable(ABC): """One named array in a CF-conforming dataset.""" name: str + location: str # the dataset's path or URL; F2 dimensions: tuple[str, ...] shape: tuple[int, ...] dtype: np.dtype size: int fill_value: Any | None chunking: tuple[int, ...] | None # None when the store is unchunked - attributes: MutableMapping[str, Any] # materialised once; tracks reads + attributes: MutableMapping[str, Any] # materialised once def __getitem__(self, keys) -> np.ndarray: ... def __setitem__(self, keys, values) -> None: ... + def __len__(self) -> int: ... # defaults to self.shape[0]; F5 + ndim: int # defaults to len(self.shape) def write_handle(self) -> Any: ... # picklable __setitem__ target; see 4.5 + def deprecated_netcdf_member(self, name: str) -> Any: ... # see 4.3 class CFDataset(ABC): """A CF-conforming array store, open for reading or writing.""" location: str # path or URL, for messages and proxies mode: str # "r" | "r+" | "a" | "w" | "w-"; 4.5 + closed: bool # F3 variables: Mapping[str, CFDatasetVariable] dimensions: Mapping[str, int] attributes: MutableMapping[str, Any] - def create_dimension(self, name: str, size: int) -> None: ... - def create_variable(self, name, dtype, dimensions, *, + def create_dimension(self, name: str, size: int | None) -> None: ... + def create_variable(self, name, dtype, dimensions=(), *, fill_value=None, **encoding) -> CFDatasetVariable: ... def sync(self) -> None: ... def finalise(self) -> None: ... # one-shot; NOT part of close() def close(self) -> None: ... + + def __enter__(self) -> "CFDataset": ... # returns self; concrete + def __exit__(self, *exc_info) -> None: ... # calls close(); concrete ``` Three members exist only to keep multi-process writing reachable later, and @@ -309,7 +317,31 @@ are explained in §4.5: `write_handle`, `mode` and `finalise`. They cost almost nothing now — `mode` is a string the implementations already track internally, `write_handle` returns `self` for Zarr, and `finalise` is a no-op for netCDF — and their absence is what would force a breaking interface -change later. +change later. `finalise` is defined but not yet called: PR 2 left +`Saver.__exit__` alone rather than guess at an ordering only Zarr can +exercise, and PR 6 wires it. + +`attributes` is materialised once and does *not* track reads. Tracking is a +`CFVariable` concern, because `CFReader` builds a second `CFVariable` over +the same backing variable when it promotes one, and the two must keep +independent read sets (§4.3). `CFDatasetVariable.attributes` is therefore a +plain mapping, and `CFVariable.attributes` is a tracking view over it. The +view's class, `TrackedAttributes`, lives in `cf/dataset.py` beside the two +abstract classes, because it is part of the same contract. + +`location` repeats `CFDataset.location` on the variable so that a variable +can name its own file without a back-reference to its dataset. It is +load-bearing: it is the `path` of a read proxy, and so part of the dask +array cache key, and it is what `LOAD_PROBLEMS.record()` reports. + +`closed` is on the interface because `Saver.complete()` has to refuse to run +until the file is released, and `isopen()` is netCDF vocabulary. A dataset +the caller opened is closed by the caller, so an implementation answers from +the store where it can, not only from its own flag. + +`create_dimension` accepts `size=None` to request an unlimited dimension; a +store with no such concept raises. `create_variable`'s `dimensions` defaults +to `()`, because grid-mapping variables are scalar. The split that matters is `attributes` versus everything else. Today `cf_var.units` might be a CF attribute or a netCDF property and the caller @@ -384,6 +416,12 @@ CF-reserved attribute silently leaks onto loaded cubes. `.attributes` is therefore a tracking mapping, not a plain `dict`, and PR 1's tests must pin `cf_attrs_unused()` before PR 2 touches it. +`CFVariable.attributes` is a `TrackedAttributes` view, constructed per +`CFVariable` over a snapshot of `cf_data.attributes`. It records reads — +which is how `cf_attrs_unused()` decides what reaches `cube.attributes` — +and it is a snapshot rather than a live view because a backend's attribute +mapping may write through to the file, and loading must never write. + ### 4.4 Reading #### Dimension names @@ -1537,7 +1575,7 @@ on the relocation before them, and PR 7 depends on everything. | # | Title | State | Link | |---|---|---|---| | 1 | `iris.fileformats.cf` becomes a package, with tests first | In review | [#7298](https://github.com/SciTools/iris/pull/7298) | -| 2 | `CFDataset`, and the CF variable classes rewritten against it | Not started | — | +| 2 | `CFDataset`, and the CF variable classes rewritten against it | In review | [#7303](https://github.com/SciTools/iris/pull/7303) | | 3 | Relocate the CF loader | Not started | — | | 4 | Zarr loading | Not started | — | | 5 | Relocate the CF saver | Not started | — | @@ -1587,6 +1625,7 @@ closing keywords live (§5). | Q5 | `iris.save(..., compute=False)` returns a tuple, not the documented `Delayed`, so `result.compute()` raises `AttributeError` **[verified]** | Caused by #6451 adapting to dask/dask#11844; the code is right and the docstrings were left behind. Not fixed here — PR 5 is behaviour-preserving. `zarr/saver.py` returns a real `Delayed` and tests it | [#7291](https://github.com/SciTools/iris/issues/7291) | Nothing | | Q6 | The `as_lazy_data` cache is process-wide and keyed on metadata, so it cannot detect a store whose chunk contents changed under identical metadata. Should it be scoped to a load session instead? | Metadata identity closes the cases that were reproduced (§4.4) and matches the netCDF key's existing strength, so it ships. Session scoping is the durable fix and would cover both formats | §4.4 | Nothing | | Q7 | Iris writes a floating `_FillValue` as base64 to stay readable by xarray, deviating from CF §2.5.1. Should the deviation be raised with the CF-Zarr conventions group rather than carried privately? | Carry it now, since the alternative is unreadable output; take it upstream to zarr-conventions/CF so the convention settles rather than each reader guessing | §4.4, §4.5 | Nothing | +| Q8 | `_add_grid_mapping_to_dataset` sets sixty-three CF grid-mapping parameters by Python attribute assignment, bypassing the ASCII-to-bytes coercion every other saved attribute goes through — so `crs_wkt` is written as `NC_STRING` while `grid_mapping_name` is `NC_CHAR`. Should they be regularised? | Not in this programme. PR 2 kept them on a named netCDF4 handle (`grid_variable`) rather than change what lands in the file. Regularising is a one-line-per-parameter change with a real CDL diff, and belongs in its own pull request | §4.5 | Nothing | Close a question by moving it to §12.4 with the date and the answer. Do not delete it. @@ -1813,6 +1852,60 @@ Every one of these is now a named test in §6. The cross-reader test that catches the third was already promised there before the review; it had simply not been written yet. +**2026-09-24 — during PR 2** + +- `CFDatasetVariable.attributes` does not track reads; `CFVariable` does. + Two `CFVariable`s can share one backing variable, and they must not share + a read set. +- `CFVariable.attributes` is a snapshot. A backend attribute mapping writes + through to the file, and a read-mode load must not write. +- The `spans` gap §5 called a latent bug is unreachable from Iris: + `_NCZARR_SCALAR_DIMENSION` only ever appears alone, so the `len == 1` + guard it lacks can never fire. PR 2 characterises the behaviour instead of + changing it, and the "one behaviour change" §5 allows is spent elsewhere. +- Saving a coordinate attribute whose name collides with a netCDF4 Python + member — `shape`, `size`, `dtype`, `name`, `dimensions`, `mask` — now + writes it, where the `hasattr()` "don't clobber" check used to drop it + silently. This is the saver-side half of the same defect the `__getattr__` + rewrite fixes on the load side. +- `finalise()` is on the interface but unwired until PR 6. +- The five declared properties — `dimensions`, `shape`, `ndim`, `dtype` and + `size` — stay typed properties on `CFVariable` and keep resolving to the + storage object. What changed is that the read is no longer *recorded*, so a + file attribute of one of those names is no longer consumed and now reaches + `cube.attributes`. A different mechanism from the bullet above, with the + same headline; the two must not be described as one. +- A name `ncattrs()` lists but `getncattr()` cannot fetch yields `""` rather + than raising, so a malformed file that used to fail now loads. The netCDF + dataset layer already shipped that leniency, and two answers for the same + malformed file depending on which layer read it is worse than one lenient + answer. +- CF attribute reads move from attribute access to a `Mapping` subscript, so + an absent name raises `KeyError` where it used to raise `AttributeError`. + Accepted as the cost of making `.attributes` the path library code takes. +- `CFReader` writes its synthesised `bounds` link into the attributes mapping, + so a formula-term or derived-bounds variable that falls back to + `build_raw_cube` carries that key into `IRIS_RAW`. Accepted rather than + filtering inside `build_raw_cube`, which reads the mapping unfiltered by + design. +- A borrowed *bare* `netCDF4.Dataset` — one that is not already an Iris + wrapper — is now wrapped in an `EncodedDataset`, so a borrow behaves like + every other input and character data decodes. The wrapping is + unconditional and does not consult `DECODE_TO_STRINGS_ON_READ`. +- A dataset that *emulates* netCDF4 must now expose three more members: + `Dimension.size`, `Dataset.ncattrs()` and `Variable.ncattrs()`. No + `getattr` fallback was added — all three are public `netCDF4` API, and + `lib/iris/AGENTS.md` bans defensive wrapping for an unconfirmed problem. + Unverified against ncdata, which is not installed in `iris-dev`. +- `cf_patch` keeps receiving netCDF4 objects. Both the dataset and the + variable handed to the hook are the netCDF4 ones, not the CF dataset + wrappers, because the hook's documented contract is netCDF4 attribute + assignment. +- Two test modules join `_PERMITTED_SUFFIXES` in + `.hooks/check_netcdf4_imports.py` — `test_NetCDFDataset.py` and + `test_CFReader__dataset.py`. Both need a genuinely bare, unwrapped + `netCDF4.Dataset`, which is precisely the input whose handling changed. + ### 12.5 Artefacts | Artefact | Location | State | @@ -1843,3 +1936,4 @@ endpoint is documented as closing on **30 September 2026** (§8). | 2026-09-22 | Covered the write proxy (§4.5): no Zarr equivalent needed, `write_handle()` justified by present netCDF need, native Zarr regains the deferred saving NCZarr gave up, and the `da.store` return-type defect recorded as Q5. | | 2026-09-23 | Accepted all five findings of the #7292 review, all reproduced. Write alignment restated over shards; fill-value masking made version-aware and taken off the storage field; the base64 `_FillValue` corrected from "malformation" to xarray's convention, and now written as well as read; the read unit separated from the write unit; the Zarr cache keyed on `Array.metadata`. Tests named in §6; Q6 and Q7 opened. | | 2026-09-23 | Structural pass for readability. §4.4 and §4.5 given `####` subheadings throughout — they were 617 lines navigated only by run-in bold lead-ins, and `Encoding` had been nested under multi-process writes by accident. Design history recast from "an earlier draft said X" into the rule it implies ("do not do X, because Y"): same guidance against re-deriving the rejected answer, without depending on knowledge of a draft the reader never saw. No normative content changed. | +| 2026-09-24 | PR 2 built. §4.2 reconciled with the implemented interface: `location`, `__len__` and `ndim` on the variable, `closed`, `__enter__` and `__exit__` on the dataset, `attributes` no longer tracking, `create_dimension(size=None)` and `create_variable(dimensions=())`. §4.3 says what `CFVariable.attributes` is. Q8 opened on the grid-mapping assignments; thirteen decisions logged. | diff --git a/lib/iris/fileformats/_nc_load_rules/actions.py b/lib/iris/fileformats/_nc_load_rules/actions.py index 5f441c55bc..d189525bc7 100644 --- a/lib/iris/fileformats/_nc_load_rules/actions.py +++ b/lib/iris/fileformats/_nc_load_rules/actions.py @@ -176,7 +176,7 @@ def action_provides_grid_mapping(engine, gridmapping_fact): (var_name,) = gridmapping_fact rule_name = "fc_provides_grid_mapping" cf_var = engine.cf_var.cf_group[var_name] - grid_mapping_type = getattr(cf_var, hh.CF_ATTR_GRID_MAPPING_NAME, None) + grid_mapping_type = cf_var.attributes.get(hh.CF_ATTR_GRID_MAPPING_NAME) succeed = True if grid_mapping_type is None: @@ -557,7 +557,7 @@ def action_all_managed_attributes(engine): iris_name = handler.iris_name matches = [] for match_name in handler.netcdf_names: - match_value = getattr(var, match_name, None) + match_value = var.attributes.get(match_name) if match_value is not None: matches.append((match_name, match_value)) @@ -623,7 +623,7 @@ def action_formula_type(engine, formula_root_fact): (var_name,) = formula_root_fact cf_var = engine.cf_var.cf_group[var_name] # cf_var.standard_name is a formula type (or we should never get here). - formula_type = getattr(cf_var, "standard_name", None) + formula_type = cf_var.attributes.get("standard_name") succeed = True if formula_type not in iris.fileformats.cf.reference_terms: succeed = False diff --git a/lib/iris/fileformats/_nc_load_rules/helpers.py b/lib/iris/fileformats/_nc_load_rules/helpers.py index 7c4810ffe7..26427d426c 100644 --- a/lib/iris/fileformats/_nc_load_rules/helpers.py +++ b/lib/iris/fileformats/_nc_load_rules/helpers.py @@ -563,8 +563,8 @@ def _add_or_capture( # best to capture objects IF possible. if attr_key is not None: captured_attr = None - with contextlib.suppress(AttributeError): - captured_attr = getattr(cf_var, attr_key) + with contextlib.suppress(KeyError): + captured_attr = cf_var.attributes[attr_key] captured = {attr_key: captured_attr} else: with contextlib.suppress(Exception): @@ -613,7 +613,7 @@ def build_raw_cube(cf_var: cf.CFVariable) -> Cube: ################################################################################ def _build_name_standard(cf_var: cf.CFVariable) -> str | None: - value = getattr(cf_var, CF_ATTR_STD_NAME, None) + value = cf_var.attributes.get(CF_ATTR_STD_NAME) if value is not None: standard_name = _get_valid_standard_name(value) else: @@ -622,7 +622,7 @@ def _build_name_standard(cf_var: cf.CFVariable) -> str | None: def _build_name_long(cf_var: cf.CFVariable) -> str | None: - return getattr(cf_var, CF_ATTR_LONG_NAME, None) + return cf_var.attributes.get(CF_ATTR_LONG_NAME) def _build_name_var(cf_var: cf.CFVariable) -> str | None: @@ -678,6 +678,13 @@ def setter(attr_name): engine.cube.attributes["invalid_standard_name"] = invalid_std_name _ = _add_or_capture( + # `_build_name_var` returns `cf_var.cf_name` - a structural member, + # not a file attribute - so the `attr_key="cf_name"` capture branch in + # `_add_or_capture` can only be reached if that plain attribute lookup + # itself raises. It doesn't today; if `_build_name_var` ever grows + # logic that can fail, this would capture `{"cf_name": None}` via + # `cf_var.attributes["cf_name"]`, which is a KeyError (cf_name is + # never in `.attributes`), not a meaningful value. build_func=partial(_build_name_var, engine.cf_var), add_method=setter("var_name"), cf_var=engine.cf_var, @@ -742,7 +749,7 @@ def build_and_add_units(engine: Engine): ################################################################################ def _build_cell_methods(cf_var: cf.CFDataVariable) -> List[iris.coords.CellMethod]: - nc_att_cell_methods = getattr(cf_var, CF_ATTR_CELL_METHODS, None) + nc_att_cell_methods = cf_var.attributes.get(CF_ATTR_CELL_METHODS) return parse_cell_methods(nc_att_cell_methods, cf_var.cf_name) @@ -771,9 +778,9 @@ def _get_ellipsoid(cf_grid_var): `cf_grid_var`. Returns None if no relevant properties are specified. """ - major = getattr(cf_grid_var, CF_ATTR_GRID_SEMI_MAJOR_AXIS, None) - minor = getattr(cf_grid_var, CF_ATTR_GRID_SEMI_MINOR_AXIS, None) - inverse_flattening = getattr(cf_grid_var, CF_ATTR_GRID_INVERSE_FLATTENING, None) + major = cf_grid_var.attributes.get(CF_ATTR_GRID_SEMI_MAJOR_AXIS) + minor = cf_grid_var.attributes.get(CF_ATTR_GRID_SEMI_MINOR_AXIS) + inverse_flattening = cf_grid_var.attributes.get(CF_ATTR_GRID_INVERSE_FLATTENING) # Avoid over-specification exception. if major is not None and minor is not None: @@ -781,12 +788,12 @@ def _get_ellipsoid(cf_grid_var): # Check for a default spherical earth. if major is None and minor is None and inverse_flattening is None: - major = getattr(cf_grid_var, CF_ATTR_GRID_EARTH_RADIUS, None) + major = cf_grid_var.attributes.get(CF_ATTR_GRID_EARTH_RADIUS) - datum = getattr(cf_grid_var, CF_ATTR_GRID_DATUM, None) + datum = cf_grid_var.attributes.get(CF_ATTR_GRID_DATUM) # Check crs_wkt if no datum if datum is None: - crs_wkt = getattr(cf_grid_var, CF_ATTR_GRID_CRS_WKT, None) + crs_wkt = cf_grid_var.attributes.get(CF_ATTR_GRID_CRS_WKT) if crs_wkt is not None: proj_crs = pyproj.crs.CRS.from_wkt(crs_wkt) if proj_crs.datum is not None: @@ -829,15 +836,17 @@ def build_rotated_coordinate_system(engine, cf_grid_var): """Create a rotated coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - north_pole_latitude = getattr(cf_grid_var, CF_ATTR_GRID_NORTH_POLE_LAT, 90.0) - north_pole_longitude = getattr(cf_grid_var, CF_ATTR_GRID_NORTH_POLE_LON, 0.0) + north_pole_latitude = cf_grid_var.attributes.get(CF_ATTR_GRID_NORTH_POLE_LAT, 90.0) + north_pole_longitude = cf_grid_var.attributes.get(CF_ATTR_GRID_NORTH_POLE_LON, 0.0) if north_pole_latitude is None or north_pole_longitude is None: warnings.warn( "Rotated pole position is not fully specified", category=iris.warnings.IrisCfLoadWarning, ) - north_pole_grid_lon = getattr(cf_grid_var, CF_ATTR_GRID_NORTH_POLE_GRID_LON, 0.0) + north_pole_grid_lon = cf_grid_var.attributes.get( + CF_ATTR_GRID_NORTH_POLE_GRID_LON, 0.0 + ) rcs = iris.coord_systems.RotatedGeogCS( north_pole_latitude, @@ -854,27 +863,27 @@ def build_transverse_mercator_coordinate_system(engine, cf_grid_var): """Create a transverse Mercator coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_central_meridian = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_CENT_MERIDIAN, None + longitude_of_central_meridian = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_CENT_MERIDIAN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) - scale_factor_at_central_meridian = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_CENT_MERIDIAN, None + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) + scale_factor_at_central_meridian = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_CENT_MERIDIAN ) # The following accounts for the inconsistency in the transverse # mercator description within the CF spec. if longitude_of_central_meridian is None: - longitude_of_central_meridian = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_central_meridian = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) if scale_factor_at_central_meridian is None: - scale_factor_at_central_meridian = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + scale_factor_at_central_meridian = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) cs = iris.coord_systems.TransverseMercator( @@ -894,15 +903,15 @@ def build_lambert_conformal_coordinate_system(engine, cf_grid_var): """Create a Lambert conformal conic coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_central_meridian = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_CENT_MERIDIAN, None + longitude_of_central_meridian = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_CENT_MERIDIAN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) - standard_parallel = getattr(cf_grid_var, CF_ATTR_GRID_STANDARD_PARALLEL, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) + standard_parallel = cf_grid_var.attributes.get(CF_ATTR_GRID_STANDARD_PARALLEL) cs = iris.coord_systems.LambertConformal( latitude_of_projection_origin, @@ -921,18 +930,18 @@ def build_stereographic_coordinate_system(engine, cf_grid_var): """Create a stereographic coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) - scale_factor_at_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + scale_factor_at_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) cs = iris.coord_systems.Stereographic( latitude_of_projection_origin, @@ -952,19 +961,19 @@ def build_polar_stereographic_coordinate_system(engine, cf_grid_var): """Create a polar stereographic coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_STRAIGHT_VERT_LON, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_STRAIGHT_VERT_LON ) - true_scale_lat = getattr(cf_grid_var, CF_ATTR_GRID_STANDARD_PARALLEL, None) - scale_factor_at_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + true_scale_lat = cf_grid_var.attributes.get(CF_ATTR_GRID_STANDARD_PARALLEL) + scale_factor_at_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) cs = iris.coord_systems.PolarStereographic( latitude_of_projection_origin, @@ -984,14 +993,14 @@ def build_mercator_coordinate_system(engine, cf_grid_var): """Create a Mercator coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) - standard_parallel = getattr(cf_grid_var, CF_ATTR_GRID_STANDARD_PARALLEL, None) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) - scale_factor_at_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + standard_parallel = cf_grid_var.attributes.get(CF_ATTR_GRID_STANDARD_PARALLEL) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) + scale_factor_at_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) cs = iris.coord_systems.Mercator( @@ -1011,14 +1020,14 @@ def build_lambert_azimuthal_equal_area_coordinate_system(engine, cf_grid_var): """Create a lambert azimuthal equal area coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) cs = iris.coord_systems.LambertAzimuthalEqualArea( latitude_of_projection_origin, @@ -1036,15 +1045,15 @@ def build_albers_equal_area_coordinate_system(engine, cf_grid_var): """Create a albers conical equal area coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_central_meridian = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_CENT_MERIDIAN, None + longitude_of_central_meridian = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_CENT_MERIDIAN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) - standard_parallels = getattr(cf_grid_var, CF_ATTR_GRID_STANDARD_PARALLEL, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) + standard_parallels = cf_grid_var.attributes.get(CF_ATTR_GRID_STANDARD_PARALLEL) cs = iris.coord_systems.AlbersEqualArea( latitude_of_projection_origin, @@ -1063,17 +1072,17 @@ def build_vertical_perspective_coordinate_system(engine, cf_grid_var): """Create a vertical perspective coordinate system from the CF-netCDF grid mapping variables.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) - perspective_point_height = getattr( - cf_grid_var, CF_ATTR_GRID_PERSPECTIVE_HEIGHT, None + perspective_point_height = cf_grid_var.attributes.get( + CF_ATTR_GRID_PERSPECTIVE_HEIGHT ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) cs = iris.coord_systems.VerticalPerspective( latitude_of_projection_origin, @@ -1092,18 +1101,18 @@ def build_geostationary_coordinate_system(engine, cf_grid_var): """Create a geostationary coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) - perspective_point_height = getattr( - cf_grid_var, CF_ATTR_GRID_PERSPECTIVE_HEIGHT, None + perspective_point_height = cf_grid_var.attributes.get( + CF_ATTR_GRID_PERSPECTIVE_HEIGHT ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) - sweep_angle_axis = getattr(cf_grid_var, CF_ATTR_GRID_SWEEP_ANGLE_AXIS, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) + sweep_angle_axis = cf_grid_var.attributes.get(CF_ATTR_GRID_SWEEP_ANGLE_AXIS) cs = iris.coord_systems.Geostationary( latitude_of_projection_origin, @@ -1123,18 +1132,18 @@ def build_oblique_mercator_coordinate_system(engine, cf_grid_var): """Create an oblique mercator coordinate system from the CF-netCDF grid mapping variable.""" ellipsoid = _get_ellipsoid(cf_grid_var) - azimuth_of_central_line = getattr(cf_grid_var, CF_ATTR_GRID_AZIMUTH_CENT_LINE, None) - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + azimuth_of_central_line = cf_grid_var.attributes.get(CF_ATTR_GRID_AZIMUTH_CENT_LINE) + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - longitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LON_OF_PROJ_ORIGIN, None + longitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LON_OF_PROJ_ORIGIN ) - scale_factor_at_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + scale_factor_at_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) - false_easting = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_EASTING, None) - false_northing = getattr(cf_grid_var, CF_ATTR_GRID_FALSE_NORTHING, None) + false_easting = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_EASTING) + false_northing = cf_grid_var.attributes.get(CF_ATTR_GRID_FALSE_NORTHING) kwargs = dict( azimuth_of_central_line=azimuth_of_central_line, latitude_of_projection_origin=latitude_of_projection_origin, @@ -1146,7 +1155,7 @@ def build_oblique_mercator_coordinate_system(engine, cf_grid_var): ) # Handle the alternative form noted in CF: rotated mercator. - grid_mapping_name = getattr(cf_grid_var, CF_ATTR_GRID_MAPPING_NAME) + grid_mapping_name = cf_grid_var.attributes[CF_ATTR_GRID_MAPPING_NAME] candidate_systems = dict( oblique_mercator=iris.coord_systems.ObliqueMercator, rotated_mercator=iris.coord_systems.RotatedMercator, @@ -1166,7 +1175,7 @@ def build_oblique_mercator_coordinate_system(engine, cf_grid_var): ################################################################################ def get_attr_units(cf_var, attributes, capture_invalid=False): - attr_units = getattr(cf_var, CF_ATTR_UNITS, UNKNOWN_UNIT_STRING) + attr_units = cf_var.attributes.get(CF_ATTR_UNITS, UNKNOWN_UNIT_STRING) if not attr_units: attr_units = UNKNOWN_UNIT_STRING @@ -1210,14 +1219,16 @@ def get_attr_units(cf_var, attributes, capture_invalid=False): attr_units = UNKNOWN_UNIT_STRING if any( - hasattr(cf_var.cf_data, name) + # Deliberately untracked: asking whether these exist must not count as + # using them, or they would stop being copied onto the cube. + name in cf_var.attributes.untracked for name in ("flag_values", "flag_masks", "flag_meanings") ): attr_units = cf_units._NO_UNIT_STRING # Get any associated calendar for a time reference coordinate. if cf_units.as_unit(attr_units).is_time_reference(): - attr_calendar = getattr(cf_var, CF_ATTR_CALENDAR, None) + attr_calendar = cf_var.attributes.get(CF_ATTR_CALENDAR) if attr_calendar: attr_units = cf_units.Unit(attr_units, calendar=attr_calendar) @@ -1228,8 +1239,8 @@ def get_attr_units(cf_var, attributes, capture_invalid=False): ################################################################################ def get_names(cf_coord_var, coord_name, attributes): """Determine the standard_name, long_name and var_name attributes.""" - standard_name = getattr(cf_coord_var, CF_ATTR_STD_NAME, None) - long_name = getattr(cf_coord_var, CF_ATTR_LONG_NAME, None) + standard_name = cf_coord_var.attributes.get(CF_ATTR_STD_NAME) + long_name = cf_coord_var.attributes.get(CF_ATTR_LONG_NAME) cf_name = str(cf_coord_var.cf_name) if standard_name is not None: @@ -1264,6 +1275,15 @@ def get_names(cf_coord_var, coord_name, attributes): ################################################################################ def get_cf_bounds_var(cf_coord_var): """Return the CF variable representing the bounds of a coordinate variable.""" + # `getattr` resolves through CFVariable.__getattr__ into `.attributes`, + # so it sees exactly what `.attributes.get(CF_ATTR_BOUNDS)` would. That + # includes the "newstyle" derived-bounds links (FUTURE.derived_bounds) + # that iris.fileformats.cf._reader synthesises: those are stored in + # `cf_var.attributes["bounds"]` like any other CF attribute, so there is + # nothing extra to look for here. _reader also *invalidates* a link by + # storing None under that key, which this call cannot tell apart from the + # default given here - fine, because both mean "no bounds", but anyone + # who comes to need the difference must subscript the mapping instead. attr_bounds = getattr(cf_coord_var, CF_ATTR_BOUNDS, None) attr_climatology = getattr(cf_coord_var, CF_ATTR_CLIMATOLOGY, None) @@ -1833,7 +1853,7 @@ def _is_lat_lon(cf_var, ud_units, std_name, std_name_grid, axis_name, prefixes): """ is_valid = False - attr_units = getattr(cf_var, CF_ATTR_UNITS, None) + attr_units = cf_var.attributes.get(CF_ATTR_UNITS) if isinstance(attr_units, str): attr_units = attr_units.lower() @@ -1841,18 +1861,18 @@ def _is_lat_lon(cf_var, ud_units, std_name, std_name_grid, axis_name, prefixes): # Special case - Check for rotated pole. if attr_units == "degrees": - attr_std_name = getattr(cf_var, CF_ATTR_STD_NAME, None) + attr_std_name = cf_var.attributes.get(CF_ATTR_STD_NAME) if attr_std_name is not None: is_valid = attr_std_name.lower() == std_name_grid else: is_valid = False # TODO: check that this interpretation of axis is correct. - attr_axis = getattr(cf_var, CF_ATTR_AXIS, None) + attr_axis = cf_var.attributes.get(CF_ATTR_AXIS) if attr_axis is not None: is_valid = attr_axis.lower() == axis_name else: # Alternative is to check standard_name or axis. - attr_std_name = getattr(cf_var, CF_ATTR_STD_NAME, None) + attr_std_name = cf_var.attributes.get(CF_ATTR_STD_NAME) if attr_std_name is not None: attr_std_name = attr_std_name.lower() @@ -1862,7 +1882,7 @@ def _is_lat_lon(cf_var, ud_units, std_name, std_name_grid, axis_name, prefixes): [attr_std_name.startswith(prefix) for prefix in prefixes] ) else: - attr_axis = getattr(cf_var, CF_ATTR_AXIS, None) + attr_axis = cf_var.attributes.get(CF_ATTR_AXIS) if attr_axis is not None: is_valid = attr_axis.lower() == axis_name @@ -1902,8 +1922,8 @@ def is_longitude(engine, cf_name): def is_projection_x_coordinate(engine, cf_name): """Determine whether the CF coordinate variable is a projection_x_coordinate variable.""" cf_var = engine.cf_var.cf_group[cf_name] - attr_name = getattr(cf_var, CF_ATTR_STD_NAME, None) or getattr( - cf_var, CF_ATTR_LONG_NAME, None + attr_name = cf_var.attributes.get(CF_ATTR_STD_NAME) or cf_var.attributes.get( + CF_ATTR_LONG_NAME ) return attr_name == CF_VALUE_STD_NAME_PROJ_X @@ -1912,8 +1932,8 @@ def is_projection_x_coordinate(engine, cf_name): def is_projection_y_coordinate(engine, cf_name): """Determine whether the CF coordinate variable is a projection_y_coordinate variable.""" cf_var = engine.cf_var.cf_group[cf_name] - attr_name = getattr(cf_var, CF_ATTR_STD_NAME, None) or getattr( - cf_var, CF_ATTR_LONG_NAME, None + attr_name = cf_var.attributes.get(CF_ATTR_STD_NAME) or cf_var.attributes.get( + CF_ATTR_LONG_NAME ) return attr_name == CF_VALUE_STD_NAME_PROJ_Y @@ -1926,10 +1946,10 @@ def is_time(engine, cf_name): """ cf_var = engine.cf_var.cf_group[cf_name] - attr_units = getattr(cf_var, CF_ATTR_UNITS, None) + attr_units = cf_var.attributes.get(CF_ATTR_UNITS) - attr_std_name = getattr(cf_var, CF_ATTR_STD_NAME, None) - attr_axis = getattr(cf_var, CF_ATTR_AXIS, "") + attr_std_name = cf_var.attributes.get(CF_ATTR_STD_NAME) + attr_axis = cf_var.attributes.get(CF_ATTR_AXIS, "") try: is_time_reference = cf_units.Unit(attr_units or 1).is_time_reference() except ValueError: @@ -1945,7 +1965,7 @@ def is_time_period(engine, cf_name): """Determine whether the CF coordinate variable represents a time period.""" is_valid = False cf_var = engine.cf_var.cf_group[cf_name] - attr_units = getattr(cf_var, CF_ATTR_UNITS, None) + attr_units = cf_var.attributes.get(CF_ATTR_UNITS) if attr_units is not None: try: @@ -1961,7 +1981,7 @@ def is_grid_mapping(engine, cf_name, grid_mapping): """Determine whether the CF grid mapping variable is of the appropriate type.""" is_valid = False cf_var = engine.cf_var.cf_group[cf_name] - attr_mapping_name = getattr(cf_var, CF_ATTR_GRID_MAPPING_NAME, None) + attr_mapping_name = cf_var.attributes.get(CF_ATTR_GRID_MAPPING_NAME) if attr_mapping_name is not None: is_valid = attr_mapping_name.lower() == grid_mapping @@ -2009,12 +2029,12 @@ def _is_rotated(engine, cf_name, cf_attr_value): """Determine whether the CF coordinate variable is rotated.""" is_valid = False cf_var = engine.cf_var.cf_group[cf_name] - attr_std_name = getattr(cf_var, CF_ATTR_STD_NAME, None) + attr_std_name = cf_var.attributes.get(CF_ATTR_STD_NAME) if attr_std_name is not None: is_valid = attr_std_name.lower() == cf_attr_value else: - attr_units = getattr(cf_var, CF_ATTR_UNITS, None) + attr_units = cf_var.attributes.get(CF_ATTR_UNITS) if attr_units is not None: is_valid = attr_units.lower() == "degrees" @@ -2043,9 +2063,9 @@ def has_supported_mercator_parameters(engine, cf_name): is_valid = True cf_grid_var = engine.cf_var.cf_group[cf_name] - standard_parallel = getattr(cf_grid_var, CF_ATTR_GRID_STANDARD_PARALLEL, None) - scale_factor_at_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + standard_parallel = cf_grid_var.attributes.get(CF_ATTR_GRID_STANDARD_PARALLEL) + scale_factor_at_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) if scale_factor_at_projection_origin is not None and standard_parallel is not None: @@ -2070,13 +2090,13 @@ def has_supported_polar_stereographic_parameters(engine, cf_name): is_valid = True cf_grid_var = engine.cf_var.cf_group[cf_name] - latitude_of_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN, None + latitude_of_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_LAT_OF_PROJ_ORIGIN ) - standard_parallel = getattr(cf_grid_var, CF_ATTR_GRID_STANDARD_PARALLEL, None) - scale_factor_at_projection_origin = getattr( - cf_grid_var, CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN, None + standard_parallel = cf_grid_var.attributes.get(CF_ATTR_GRID_STANDARD_PARALLEL) + scale_factor_at_projection_origin = cf_grid_var.attributes.get( + CF_ATTR_GRID_SCALE_FACTOR_AT_PROJ_ORIGIN ) if latitude_of_projection_origin != 90 and latitude_of_projection_origin != -90: diff --git a/lib/iris/fileformats/cf/_reader.py b/lib/iris/fileformats/cf/_reader.py index 7b33103af5..5bda587a5b 100644 --- a/lib/iris/fileformats/cf/_reader.py +++ b/lib/iris/fileformats/cf/_reader.py @@ -40,13 +40,15 @@ ``__init__`` fails part way. This is the one module in :mod:`iris.fileformats.cf` still coupled to -:mod:`iris.fileformats.netcdf`: opening a file needs ``_thread_safe_nc`` and -``_bytecoding_datasets``. Everything else in the package already works against -any object that presents netCDF-like ``variables``, ``dimensions`` and -``ncattrs``. Removing that last coupling -- replacing the direct dataset -construction with a format-agnostic dataset abstraction, so that Zarr can be -read by the same machinery -- is the subject of §4.2 of the native Zarr I/O -design, ``docs/superpowers/specs/2026-09-21-zarr-io-design.md``. +:mod:`iris.fileformats.netcdf`: opening a file means building a +:class:`~iris.fileformats.netcdf._dataset.NetCDFDataset`. Everything else in +the package works against the format-agnostic +:class:`~iris.fileformats.cf.dataset.CFDataset` / +:class:`~iris.fileformats.cf.dataset.CFDatasetVariable` interface, so a Zarr +store can be read by the same machinery once a +:class:`~iris.fileformats.cf.dataset.CFDataset` implementation exists for it +-- see §4.2 of the native Zarr I/O design, +``docs/superpowers/specs/2026-09-21-zarr-io-design.md``. """ @@ -58,7 +60,7 @@ import iris.exceptions import iris.fileformats._nc_load_rules.helpers as hh -from iris.fileformats.netcdf import _bytecoding_datasets, _thread_safe_nc +from iris.fileformats.netcdf._dataset import NetCDFDataset import iris.warnings from ._group import CFGroup @@ -139,18 +141,17 @@ def __init__(self, file_source, warn=False, monotonic=False): else: self._filename = file_source - if _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ: - ds_type = _bytecoding_datasets.EncodedDataset - else: - ds_type = _thread_safe_nc.DatasetWrapper - - self._dataset = ds_type(self._filename, mode="r") + self._dataset = NetCDFDataset( + self._filename, mode="r", warn_legacy_format=warn + ) self._own_file = True else: # We have been passed an open dataset. # We use it but don't own it (don't close it). - self._dataset = file_source - self._filename = self._dataset.filepath() + self._dataset = NetCDFDataset.from_existing( + file_source, warn_legacy_format=warn + ) + self._filename = self._dataset.location #: Collection of CF-netCDF variables associated with this netCDF file self.cf_group = self.CFGroup() @@ -158,17 +159,6 @@ def __init__(self, file_source, warn=False, monotonic=False): # Result of parsing "grid_mapping" attribute; mapping of coordinate_system => coordinates self._coord_system_mappings = {} - # Issue load optimisation warning. - if warn and self._dataset.file_format in [ - "NETCDF3_CLASSIC", - "NETCDF3_64BIT", - ]: - warnings.warn( - "Optimise CF-netCDF loading by converting data from NetCDF3 " - 'to NetCDF4 file format using the "nccopy" command.', - category=iris.warnings.IrisLoadWarning, - ) - self._check_monotonic = monotonic self._with_ugrid = True @@ -177,9 +167,8 @@ def __init__(self, file_source, warn=False, monotonic=False): self._with_ugrid = False # Read the variables in the dataset only once to reduce runtime. - ds = self._dataset - # Turn off *any* automatic decoding in the underlying netCDF4 dataset. - ds.set_auto_chartostring(False) + # NetCDFDataset.variables caches, so this is the same dict - and the + # same CFDatasetVariable objects - that _has_meshes just walked. variables = self._dataset.variables self._translate(variables) self._build_cf_groups(variables) @@ -201,7 +190,8 @@ def __exit__(self, exc_type, exc_value, traceback): def _has_meshes(self): result = False for variable in self._dataset.variables.values(): - if hasattr(variable, "mesh") or hasattr(variable, "node_coordinates"): + attributes = variable.attributes + if "mesh" in attributes or "node_coordinates" in attributes: result = True break return result @@ -232,7 +222,7 @@ def _translate(self, variables): # Parse all instances of "grid_mapping" attributes and store in CFReader # This avoids re-parsing the grid_mappings each time they are needed. for nc_var in variables.values(): - if grid_mapping_attr := getattr(nc_var, "grid_mapping", None): + if grid_mapping_attr := nc_var.attributes.get("grid_mapping"): try: cs_mappings = hh._parse_extended_grid_mapping(grid_mapping_attr) self._coord_system_mappings[nc_var.name] = cs_mappings @@ -270,11 +260,7 @@ def _translate(self, variables): ) # Identify global netCDF attributes. - attr_dict = { - attr_name: _getncattr(self._dataset, attr_name, "") - for attr_name in self._dataset.ncattrs() - } - self.cf_group.global_attributes.update(attr_dict) + self.cf_group.global_attributes.update(dict(self._dataset.attributes)) # Identify and register all CF formula terms. formula_terms = _CFFormulaTermsVariable.identify(variables) @@ -301,12 +287,14 @@ def _translate(self, variables): if cf_root_coord is None: cf_root_coord = self.cf_group.auxiliary_coordinates.get(cf_root) - root_bounds_name = getattr(cf_root_coord, "bounds", None) # N.B. cf_root_coord may here be None, if the root var was not a # coord - that is ok, it will not have a 'bounds', we will skip it. + root_bounds_name = None + if cf_root_coord is not None: + root_bounds_name = cf_root_coord.attributes.get("bounds") if root_bounds_name in self.cf_group: root_bounds_var = self.cf_group.get(root_bounds_name) - if not hasattr(root_bounds_var, "formula_terms"): + if "formula_terms" not in root_bounds_var.attributes: # this is an invalid root bounds, according to CF, and therefore should be promoted into a cube root_bounds_var._to_be_promoted = True else: @@ -322,7 +310,9 @@ def _translate(self, variables): (term_bounds_var,) = term_bounds_vars # N.B. bounds==main-var is valid CF for *no* bounds if term_bounds_var != cf_var: - cf_var.bounds = term_bounds_var.cf_name + cf_var.attributes["bounds"] = ( + term_bounds_var.cf_name + ) new_var = CFBoundaryVariable( term_bounds_var.cf_name, term_bounds_var.cf_data ) @@ -336,9 +326,9 @@ def _translate(self, variables): if cf_name not in self.cf_group: # If the formula term variable is not already in the group, add it as a coordinate. new_var = CFAuxiliaryCoordinateVariable(cf_name, cf_var.cf_data) - if iris.FUTURE.derived_bounds and hasattr(cf_var, "bounds"): + if iris.FUTURE.derived_bounds and "bounds" in cf_var.attributes: # Copy "old-style" derived bounds link - new_var.bounds = cf_var.bounds + new_var.attributes["bounds"] = cf_var.attributes["bounds"] self.cf_group[cf_name] = new_var self.cf_group[cf_name].add_formula_term(cf_root, cf_term) @@ -347,14 +337,16 @@ def _translate(self, variables): for cf_root in all_roots: # Invalidate "broken" bounds connections root_var = self.cf_group[cf_root] - if getattr(root_var, "formula_terms", None) and getattr( - root_var, "bounds", None + if root_var.attributes.get("formula_terms") and root_var.attributes.get( + "bounds" ): - root_bounds_var = self.cf_group.get(root_var.bounds) - if not getattr(root_bounds_var, "formula_terms", None): + root_bounds_var = self.cf_group.get(root_var.attributes["bounds"]) + if root_bounds_var is None or not root_bounds_var.attributes.get( + "formula_terms" + ): # This means it is *not* a valid bounds var, according to CF, and so therefore we are # invalidating the bounds. - root_var.bounds = None + root_var.attributes["bounds"] = None # Determine the CF data variables. data_variable_names = ( @@ -436,13 +428,14 @@ def _span_check( if iris.FUTURE.derived_bounds: # Include bounds of every variable, within cf_group attached to the variable. - if hasattr(cf_variable, "bounds"): - if cf_variable.bounds not in cf_group: - bounds_var = self.cf_group.get(cf_variable.bounds) + if "bounds" in cf_variable.attributes: + bounds_name = cf_variable.attributes["bounds"] + if bounds_name not in cf_group: + bounds_var = self.cf_group.get(bounds_name) if bounds_var: # TODO: warning if span fails if bounds_var.spans(cf_variable): - cf_group[cf_variable.bounds] = bounds_var + cf_group[bounds_name] = bounds_var # Build CF data variable relationships. if isinstance(cf_variable, CFDataVariable): @@ -457,7 +450,7 @@ def _span_check( } ) # Add appropriate "dimensionless" CF coordinate variables. - coordinates_attr = getattr(cf_variable, "coordinates", "") + coordinates_attr = cf_variable.attributes.get("coordinates", "") cf_group.update( { cf_name: self.cf_group[cf_name] @@ -494,9 +487,12 @@ def _span_check( for cf_root, cf_term in cf_var.cf_terms_by_root.items(): cf_root_var = self.cf_group[cf_root] if iris.FUTURE.derived_bounds: - if not hasattr(cf_root_var, "standard_name"): + if "standard_name" not in cf_root_var.attributes: continue - name = cf_root_var.standard_name or cf_root_var.long_name + name = ( + cf_root_var.attributes["standard_name"] + or cf_root_var.attributes["long_name"] + ) terms = reference_terms.get(name, []) if isinstance(terms, str) or not isinstance(terms, Iterable): terms = [terms] @@ -538,12 +534,3 @@ def _close(self): def __del__(self): # Be sure to close dataset when CFReader is destroyed / garbage-collected. self._close() - - -def _getncattr(dataset, attr, default=None): - """Wrap `netCDF4.Dataset.getncattr` to make it behave more like `getattr`.""" - try: - value = dataset.getncattr(attr) - except AttributeError: - value = default - return value diff --git a/lib/iris/fileformats/cf/_variables.py b/lib/iris/fileformats/cf/_variables.py index e4067ed584..8f547ff57a 100644 --- a/lib/iris/fileformats/cf/_variables.py +++ b/lib/iris/fileformats/cf/_variables.py @@ -15,8 +15,8 @@ attribute name, or ``cf_identities``, a list of them for the classes that answer to several -- and implements ``identify``. ``identify`` is a classmethod, not an instance method: it is called on the class, is handed the -whole ``{name: netCDF4.Variable}`` mapping of the file, and returns the subset -of it that belongs to that class, as a ``{name: CFVariable instance}`` +whole ``{name: CFDatasetVariable}`` mapping of the file, and returns the +subset of it that belongs to that class, as a ``{name: CFVariable instance}`` mapping. ``ignore`` and ``target`` narrow what it looks at; ``warn`` gates whether it complains. @@ -52,6 +52,7 @@ import numpy as np import numpy.ma as ma +from iris.fileformats.cf.dataset import TrackedAttributes from iris.mesh.components import Connectivity import iris.util import iris.warnings @@ -78,6 +79,15 @@ # therefore automatically classed as "used" attributes. _CF_ATTRS_IGNORE = set(["_FillValue", "add_offset", "missing_value", "scale_factor"]) +# Names __getattr__ must never resolve through self.attributes, because +# reading self.attributes is how it resolves anything at all. Just those two, +# and deliberately so: every other name that reaches __getattr__ is a CF +# attribute name read from a file, which is the open-ended set this class +# exists to present. Excluding a broader class of name - "every +# underscore-prefixed name", say - would hide real CF attributes, "_Encoding" +# and "_FillValue" among them. +_GETATTR_RECURSION_GUARD = frozenset(["attributes", "cf_data"]) + # NetCDF returns a different type for strings depending on Python version. def _is_str_dtype(var): @@ -95,22 +105,42 @@ class CFVariable(metaclass=ABCMeta): cf_identity: ClassVar[str | None] = None def __init__(self, name, data): - # Accessing the list of netCDF attributes is surprisingly slow. - # Since it's used repeatedly, caching the list makes things - # quite a bit faster. - self._nc_attrs = data.ncattrs() - self.cf_name = name """NetCDF variable name.""" self.cf_data = data - """NetCDF4 Variable data instance.""" + """The variable's storage. + + A :class:`~iris.fileformats.cf.dataset.CFDatasetVariable`. + ``_deprecated_netcdf_member`` is the only netCDF4-specific route out + of this class; everything else goes through the format-agnostic + interface. + """ + + self.attributes = TrackedAttributes( + dict(data.attributes), ignored=_CF_ATTRS_IGNORE + ) + """The variable's CF attributes, and a record of which have been read. + + The only place CF attributes live. ``__getattr__`` forwards here, so + ``cf_var.units`` and ``cf_var.attributes["units"]`` are the same read + and count once. What was read decides what survives onto the loaded + cube - see :func:`iris.fileformats.netcdf.loader._add_unused_attributes`. + + Copied from ``data.attributes``, not a view onto it: + :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable`'s + mapping writes through to the file, and a plain ``dict`` is what + keeps two :class:`CFVariable` built over the same storage object + from sharing one mapping. + """ - """File source of the NetCDF content.""" try: - self.filename = data.group().filepath() + location = data.location except AttributeError: - self.filename = "" + location = "" + + self.filename = location + """File source of the NetCDF content.""" self.cf_group = None """Collection of CF-netCDF variables associated with this variable.""" @@ -120,8 +150,6 @@ def __init__(self, name, data): self._to_be_promoted = False - self.cf_attrs_reset() - @staticmethod def _identify_common(variables, ignore, target): if ignore is None: @@ -147,7 +175,7 @@ def identify(self, variables, ignore=None, target=None, warn=True): Parameters ---------- variables : - Dictionary of netCDF4.Variable instance by variable name. + Dictionary of CFDatasetVariable instance by variable name. ignore : optional List of variable names to ignore. target : optional @@ -162,6 +190,16 @@ def identify(self, variables, ignore=None, target=None, warn=True): """ pass + def _is_scalar(self) -> bool: + """Whether this variable is scalar, in either spelling of scalar. + + NetCDF gives a zero-dimensional variable no dimensions at all. + NCZarr gives it one, named ``_scalar_``. The two mean the same + thing, and every ``spans`` implementation has to treat them alike. + + """ + return not self.dimensions or self.dimensions == (_NCZARR_SCALAR_DIMENSION,) + def spans(self, cf_variable): """Determine dimensionality coverage. @@ -181,11 +219,11 @@ def spans(self, cf_variable): bool """ - dimensions = tuple(self.dimensions) - if dimensions == (_NCZARR_SCALAR_DIMENSION,): + # Scalar variables always span the target variable. + if self._is_scalar(): return True - result = set(dimensions).issubset(cf_variable.dimensions) + result = set(self.dimensions).issubset(cf_variable.dimensions) return result def __eq__(self, other): @@ -200,15 +238,65 @@ def __hash__(self): # CF variable names are unique. return hash(self.cf_name) + @property + def dimensions(self) -> tuple: + """The names of the dimensions this variable spans, in order.""" + return tuple(self.cf_data.dimensions) + + @property + def shape(self) -> tuple: + """The variable's shape.""" + return self.cf_data.shape + + @property + def ndim(self) -> int: + """The number of dimensions this variable spans.""" + return len(self.shape) + + @property + def dtype(self): + """The variable's stored data type.""" + return self.cf_data.dtype + + @property + def size(self) -> int: + """The total number of elements in the variable.""" + return self.cf_data.size + def __getattr__(self, name): - # Accessing netCDF attributes is surprisingly slow. Since - # they're often read repeatedly, caching the values makes things - # quite a bit faster. - if name in self._nc_attrs: - self._cf_attrs.add(name) - value = getattr(self.cf_data, name) - setattr(self, name, value) - return value + """Return the named CF attribute, as read from the file. + + The open-ended half of this class. CF attribute names are data read + from a file, not API, so they cannot be declared - which is what + justifies ``__getattr__`` here at all. It resolves only against + :attr:`attributes`; everything structural is a declared property + above. Reading through it records the attribute as used, exactly as + ``cf_var.attributes[name]`` does. + + """ + if name.startswith("__") or name in _GETATTR_RECURSION_GUARD: + # Dunder probes - copy, pickle, numpy protocols - must not be + # answered from file data, and the two members this method reads + # must not be resolved by this method. + raise AttributeError(name) + + try: + return self.attributes[name] + except KeyError: + pass + return self._deprecated_netcdf_member(name) + + def _deprecated_netcdf_member(self, name): + """Return a netCDF4 member of the backing variable, as this class used to. + + The one-cycle compatibility route for code that reached netCDF4 API + through a CFVariable. Records nothing: this is not a CF attribute. + The storage object decides what this means, and warns - see + :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable`'s + ``deprecated_netcdf_member``. + + """ + return self.cf_data.deprecated_netcdf_member(name) def __getitem__(self, key): return self.cf_data.__getitem__(key) @@ -223,31 +311,37 @@ def __repr__(self): self.cf_data, ) + # Every cf_attrs_* reader below takes its values from + # ``self.attributes.untracked``: a report on what was read must not itself + # count as reading. Each binds that view to a local first, because reading + # the property copies the whole mapping. + def cf_attrs(self): """Return a list of all attribute name and value pairs of the CF-netCDF variable.""" - return tuple((attr, self.getncattr(attr)) for attr in sorted(self._nc_attrs)) + attributes = self.attributes.untracked + return tuple((name, attributes[name]) for name in sorted(attributes)) def cf_attrs_ignored(self): """Return a list of all ignored attribute name and value pairs of the CF-netCDF variable.""" - return tuple( - (attr, self.getncattr(attr)) - for attr in sorted(set(self._nc_attrs) & _CF_ATTRS_IGNORE) - ) + attributes = self.attributes.untracked + names = set(attributes) & _CF_ATTRS_IGNORE + return tuple((name, attributes[name]) for name in sorted(names)) def cf_attrs_used(self): """Return a list of all accessed attribute name and value pairs of the CF-netCDF variable.""" - return tuple((attr, self.getncattr(attr)) for attr in sorted(self._cf_attrs)) + attributes = self.attributes.untracked + return tuple((name, attributes[name]) for name in sorted(self.attributes.read)) def cf_attrs_unused(self): """Return a list of all non-accessed attribute name and value pairs of the CF-netCDF variable.""" + attributes = self.attributes.untracked return tuple( - (attr, self.getncattr(attr)) - for attr in sorted(set(self._nc_attrs) - self._cf_attrs) + (name, attributes[name]) for name in sorted(self.attributes.unread) ) def cf_attrs_reset(self): """Reset the history of accessed attribute names of the CF-netCDF variable.""" - self._cf_attrs = set([item[0] for item in self.cf_attrs_ignored()]) + self.attributes.reset() def add_formula_term(self, root, term): """Register the participation of this CF-netCDF variable in a CF-netCDF formula term. @@ -301,7 +395,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF ancillary data variables. for nc_var_name, nc_var in target.items(): # Check for ancillary data variable references. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: for name in nc_var_att.split(): @@ -350,7 +444,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF auxiliary coordinate variables. for nc_var_name, nc_var in target.items(): # Check for auxiliary coordinate variable references. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: for name in nc_var_att.split(): @@ -399,7 +493,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF boundary variables. for nc_var_name, nc_var in target.items(): # Check for a boundary variable reference. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: name = nc_var_att.strip() @@ -438,7 +532,7 @@ def spans(self, cf_variable): """ # Scalar variables always span the target variable. result = True - if self.dimensions: + if not self._is_scalar(): source = self.dimensions target = cf_variable.dimensions # Ignore the bounds extent dimension. @@ -475,7 +569,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF climatology variables. for nc_var_name, nc_var in target.items(): # Check for a climatology variable reference. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: name = nc_var_att.strip() @@ -514,7 +608,7 @@ def spans(self, cf_variable): """ # Scalar variables always span the target variable. result = True - if self.dimensions: + if not self._is_scalar(): source = self.dimensions target = cf_variable.dimensions # Ignore the climatology extent dimension. @@ -611,7 +705,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF formula terms variables. for nc_var_name, nc_var in target.items(): # Check for formula terms variable references. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: for match_item in _CF_PARSE.finditer(nc_var_att): @@ -684,7 +778,7 @@ def identify( # Identify all grid mapping variables. for nc_var_name, nc_var in target.items(): # Check for a grid mapping variable reference. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: # All `grid_mapping` attributes will already have been parsed prior @@ -768,7 +862,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF label variables. for nc_var_name, nc_var in target.items(): # Check for label variable references. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: for name in nc_var_att.split(): @@ -837,7 +931,7 @@ def spans(self, cf_variable): """ # Scalar variables always span the target variable. result = True - if self.dimensions: + if not self._is_scalar(): source = self.dimensions target = cf_variable.dimensions # Ignore label string length dimension. @@ -871,7 +965,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Identify all CF measure variables. for nc_var_name, nc_var in target.items(): # Check for measure variable references. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: for match_item in _CF_PARSE.finditer(nc_var_att): @@ -935,7 +1029,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # Check for connectivity variable references, iterating through # the valid cf roles. for identity in cls.cf_identities: - nc_var_att = getattr(nc_var, identity, None) + nc_var_att = nc_var.attributes.get(identity) if nc_var_att is not None: # UGRID only allows for one of each connectivity cf role. @@ -1010,7 +1104,7 @@ def identify(cls, variables, ignore=None, target=None, warn=True): for nc_var_name, nc_var in target.items(): # Check for UGRID auxiliary coordinate variable references. for identity in cls.cf_identities: - nc_var_att = getattr(nc_var, identity, None) + nc_var_att = nc_var.attributes.get(identity) if nc_var_att is not None: for name in nc_var_att.split(): @@ -1084,11 +1178,11 @@ def identify(cls, variables, ignore=None, target=None, warn=True): # SPECIAL BEHAVIOUR FOR MESH VARIABLES. # We are looking for all mesh variables. Check if THIS variable # is a mesh using its own attributes. - if getattr(nc_var, "cf_role", "") == "mesh_topology": + if nc_var.attributes.get("cf_role", "") == "mesh_topology": result[nc_var_name] = CFUGridMeshVariable(nc_var_name, nc_var) # Check for mesh variable references. - nc_var_att = getattr(nc_var, cls.cf_identity, None) + nc_var_att = nc_var.attributes.get(cls.cf_identity) if nc_var_att is not None: # UGRID only allows for 1 mesh per variable. diff --git a/lib/iris/fileformats/cf/dataset.py b/lib/iris/fileformats/cf/dataset.py new file mode 100644 index 0000000000..2d2802626f --- /dev/null +++ b/lib/iris/fileformats/cf/dataset.py @@ -0,0 +1,324 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""The format-agnostic description of a CF-conforming array store. + +.. z_reference:: iris.fileformats.cf.dataset + :tags: topic_load_save + + API reference + +:class:`CFDataset` and :class:`CFDatasetVariable` are what the rest of the CF +layer is written against. They are deliberately small: only what the +:class:`~iris.fileformats.cf.CFVariable` classes, the CF loader and the CF +saver actually need, with one implementation per storage format +(:mod:`iris.fileformats.netcdf._dataset` today, Zarr next). + +The split that matters is ``attributes`` versus everything else. A netCDF +variable presents ``units`` - a CF attribute read from the file - and +``dimensions`` - a property of the storage - through the same attribute syntax, +and a caller cannot tell which is which. Here, CF attributes are only ever +reached through ``attributes``, and storage properties are named, typed +members. + +``attributes`` is an open-ended set of data keys read from a file, with no +schema to enumerate, which is why :class:`TrackedAttributes` exists and why +:meth:`~iris.fileformats.cf.CFVariable.__getattr__` is allowed to forward to +it. Tracking is the load-bearing part: Iris decides which file attributes +survive onto a loaded cube by asking which ones the loading rules did *not* +read, so a read that goes unrecorded silently leaks a CF-reserved attribute +onto the cube. + +See sections 4.2, 4.3 and 4.5 of +``docs/superpowers/specs/2026-09-21-zarr-io-design.md``. + +""" + +from abc import ABC, abstractmethod +from collections.abc import ( + ItemsView, + Iterable, + Iterator, + KeysView, + Mapping, + MutableMapping, + ValuesView, +) +from types import MappingProxyType +from typing import Any + +import numpy as np + + +class TrackedAttributes(MutableMapping): + """A variable's attributes, recording which of them have been looked up. + + Single-key lookups record: :meth:`__getitem__`, :meth:`get` and + :meth:`__contains__` - the last because ``hasattr(cf_var, name)`` marks an + attribute used today, by reaching ``__getattr__``. Bulk access does not: + iteration, :meth:`keys`, :meth:`values`, :meth:`items` and :func:`len` + leave the record alone, and :attr:`untracked` is the explicit escape hatch + for a single-key lookup that must not count. + + """ + + def __init__(self, source: MutableMapping, *, ignored: Iterable[str] = ()): + """Wrap ``source``, treating any ``ignored`` name it holds as already read.""" + self._source = source + self._ignored = frozenset(ignored) + self._read: set[str] = set() + self.reset() + + def __getitem__(self, key: str) -> Any: + """Return ``key``'s value, recording it as read.""" + # Index first: a KeyError must escape without recording anything. + value = self._source[key] + self._read.add(key) + return value + + def __contains__(self, key: object) -> bool: + """Return whether ``key`` is present, recording a hit as read.""" + result = key in self._source + if result: + self._read.add(key) # type: ignore[arg-type] + return result + + def __setitem__(self, key: str, value: Any) -> None: + """Set ``key``'s value, recording nothing.""" + self._source[key] = value + + def __delitem__(self, key: str) -> None: + """Remove ``key``, and any record of it having been read.""" + del self._source[key] + self._read.discard(key) + + def __iter__(self) -> Iterator[str]: + """Return an iterator over the attribute names, recording nothing.""" + return iter(self._source) + + def __len__(self) -> int: + """Return the number of attributes, recording nothing.""" + return len(self._source) + + def keys(self) -> KeysView: + """Return a view of the attribute names, recording nothing.""" + return self._source.keys() + + def values(self) -> ValuesView: + """Return a view of the attribute values, recording nothing.""" + return self._source.values() + + def items(self) -> ItemsView: + """Return a view of the attribute pairs, recording nothing.""" + return self._source.items() + + def __repr__(self) -> str: + """Return a string representation, recording nothing.""" + return f"{self.__class__.__name__}({dict(self._source)!r})" + + @property + def untracked(self) -> Mapping: + """A read-only view of the same attributes that records nothing.""" + return MappingProxyType(dict(self._source)) + + @property + def read(self) -> frozenset: + """The names looked up since construction or the last :meth:`reset`.""" + return frozenset(self._read) + + @property + def unread(self) -> frozenset: + """The names present but not looked up since the last :meth:`reset`.""" + return frozenset(self._source) - self._read + + def reset(self) -> None: + """Forget every recorded lookup, re-seeding from the ignored names.""" + self._read = set(self._ignored) & set(self._source) + + +class CFDatasetVariable(ABC): + """One named array in a CF-conforming dataset.""" + + @property + @abstractmethod + def name(self) -> str: + """The variable's name within its dataset.""" + + @property + @abstractmethod + def location(self) -> str: + """The path or URL of the dataset holding this variable. + + Repeated from :attr:`CFDataset.location` with the same meaning, so that + a variable can label a message or a data proxy without a reference back + to its dataset. + + """ + + @property + @abstractmethod + def dimensions(self) -> tuple: + """The names of the dimensions this variable spans, in order.""" + + @property + @abstractmethod + def shape(self) -> tuple: + """The variable's shape.""" + + @property + @abstractmethod + def dtype(self) -> np.dtype: + """The variable's stored data type. + + Usually a :class:`numpy.dtype`. netCDF's variable-length string type + reports the builtin :class:`str` instead, and + :func:`iris.fileformats.netcdf.loader._get_cf_var_data` branches on + that, so it is passed through rather than normalised. + + """ + + @property + @abstractmethod + def size(self) -> int: + """The total number of elements in the variable.""" + + @property + @abstractmethod + def fill_value(self) -> Any: + """The store's own fill value, or ``None`` if it has none.""" + + @property + @abstractmethod + def chunking(self) -> tuple | None: + """The variable's storage chunk shape, or ``None`` when unchunked.""" + + @property + @abstractmethod + def attributes(self) -> MutableMapping: + """The variable's CF attributes, materialised once at construction.""" + + @abstractmethod + def __getitem__(self, keys) -> np.ndarray: + """Return the indexed portion of the variable's data.""" + + @abstractmethod + def __setitem__(self, keys, values) -> None: + """Write ``values`` into the indexed portion of the variable.""" + + @abstractmethod + def write_handle(self) -> Any: + """Return a picklable object supporting ``__setitem__`` on this variable. + + This is what a Dask worker receives as a ``da.store`` target, so it must + survive pickling and must remain usable after the dataset it came from + has been closed. + + """ + + @property + def ndim(self) -> int: + """The number of dimensions the variable spans.""" + return len(self.shape) + + def __len__(self) -> int: + """Return the length of the variable's leading dimension.""" + if not self.shape: + # netCDF4.Variable.__len__ raises exactly this for a scalar, and + # CFVariable.__len__ forwards to it today. + msg = "len() of unsized object" + raise TypeError(msg) + return self.shape[0] + + def deprecated_netcdf_member(self, name: str) -> Any: + """Return a netCDF4-only member of the object backing this variable. + + The one-release compatibility route for code that reached netCDF4 API + through a :class:`~iris.fileformats.cf.CFVariable`. A store with no + backing netCDF4 object has nothing to offer and raises, which is the + correct answer rather than a special case. + + """ + raise AttributeError(name) + + +class CFDataset(ABC): + """A CF-conforming array store, open for reading or writing.""" + + @property + @abstractmethod + def location(self) -> str: + """The store's path or URL, for messages and data proxies.""" + + @property + @abstractmethod + def mode(self) -> str: + """The mode the store was opened in: ``r``, ``r+``, ``a``, ``w`` or ``w-``.""" + + @property + @abstractmethod + def closed(self) -> bool: + """Whether :meth:`close` has already released the store.""" + + @property + @abstractmethod + def variables(self) -> Mapping: + """The store's variables, by name.""" + + @property + @abstractmethod + def dimensions(self) -> Mapping: + """The store's dimension lengths, by name.""" + + @property + @abstractmethod + def attributes(self) -> MutableMapping: + """The store's global attributes.""" + + @abstractmethod + def create_dimension(self, name: str, size: int | None) -> None: + """Declare a dimension of the given length. + + ``size=None`` requests an unlimited dimension. A store with no such + concept raises :class:`NotImplementedError`. + + """ + + @abstractmethod + def create_variable( + self, name: str, dtype, dimensions=(), *, fill_value=None, **encoding + ) -> "CFDatasetVariable": + """Create and return a new variable. + + ``**encoding`` is storage-specific by nature - ``zlib``, ``complevel`` + and ``chunksizes`` for netCDF; ``compressors``, ``chunks`` and + ``shards`` for Zarr - so each implementation documents the keys it + accepts, and generic CF code never constructs them. + + """ + + @abstractmethod + def sync(self) -> None: + """Flush buffered writes to the store.""" + + @abstractmethod + def finalise(self) -> None: + """Perform the store's one-shot completion step, after every write. + + Deliberately **not** part of :meth:`close`: a worker that writes one + slab of a store must close its handle without performing a completion + step that only one process may perform. + + """ + + @abstractmethod + def close(self) -> None: + """Release the store's resources.""" + + def __enter__(self) -> "CFDataset": + """Return this dataset, for use as a context manager.""" + return self + + def __exit__(self, *exc_info) -> None: + """Close this dataset on leaving the context, whatever happened in it.""" + self.close() diff --git a/lib/iris/fileformats/netcdf/_dataset.py b/lib/iris/fileformats/netcdf/_dataset.py new file mode 100644 index 0000000000..114b50d96c --- /dev/null +++ b/lib/iris/fileformats/netcdf/_dataset.py @@ -0,0 +1,526 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""The netCDF implementation of the :mod:`iris.fileformats.cf.dataset` interface. + +This is the only module that knows both the CF interface and the netCDF4 API. +Everything netCDF-specific that the CF layer used to reach through a +:class:`~iris.fileformats.cf.CFVariable` - ``ncattrs``, ``getncattr``, +``setncattr``, ``chunking()``, ``file_format``, ``createVariable``, +``createDimension``, ``filepath()``, ``isopen()`` - either has a named member +on the interface or lives here as a netCDF-only member. + +Reading and writing share one class, and the read path must not touch +:attr:`NetCDFDataset.write_lock`: making the lock raises for Dask schedulers +that only the saver cares about, so it is made on first use rather than at +construction. :meth:`NetCDFDataset.from_existing` serves both paths and +cannot tell them apart, so it stipulates an open mode rather than observing +one, and always wraps for byte encoding. + +""" + +from collections.abc import Iterator, Mapping, MutableMapping +from typing import Any +import warnings + +import numpy as np + +from iris._deprecation import warn_deprecated +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable +import iris.warnings + +from . import _bytecoding_datasets, _dask_locks, _thread_safe_nc + +#: What ``netCDF4.Variable.chunking()`` answers for an unchunked variable. +#: ``None`` means the same thing, and arrives from non-version-4 files. +_CONTIGUOUS = "contiguous" + +#: The member an ncdata emulating variable carries instead of file storage. +_EMULATED_DATA_ARRAY = "_data_array" + + +def _bytes_if_ascii(value): + """Return an ASCII string as bytes, and anything else unchanged. + + netCDF4 stores a bytes value as NC_CHAR and a str value as NC_STRING. + Iris writes its string attributes as NC_CHAR, so they are encoded here + before being set. + + """ + if isinstance(value, str): + try: + return value.encode(encoding="ascii") + except (AttributeError, UnicodeEncodeError): + pass + return value + + +class _NetCDFAttributes(MutableMapping): + """A netCDF object's attributes as a mapping: read once, written through.""" + + def __init__(self, target): + """Materialise the attributes of ``target``, a netCDF variable or dataset.""" + self._target = target + self._values: dict[str, Any] = {} + for name in target.ncattrs(): + try: + value = target.getncattr(name) + except AttributeError: + # For some malformed files, netCDF4 lists an attribute name + # that it then refuses to return. + value = "" + self._values[name] = value + + def __getitem__(self, key: str) -> Any: + """Return ``key``'s value.""" + return self._values[key] + + def __setitem__(self, key: str, value: Any) -> None: + """Set ``key``'s value, here and in the netCDF object.""" + # Cache the original, not the coerced value: a just-written attribute + # must read back as str, exactly as one read from a file does. + self._target.setncattr(key, _bytes_if_ascii(value)) + self._values[key] = value + + def __delitem__(self, key: str) -> None: + """Remove ``key``, here and from the netCDF object.""" + self._target.delncattr(key) + del self._values[key] + + def __iter__(self) -> Iterator[str]: + """Return an iterator over the attribute names, in file order.""" + return iter(self._values) + + def __len__(self) -> int: + """Return the number of attributes.""" + return len(self._values) + + def __repr__(self) -> str: + """Return a string representation.""" + return f"{self.__class__.__name__}({self._values!r})" + + +class NetCDFDatasetVariable(CFDatasetVariable): + """One variable of a netCDF file, presented through the CF interface.""" + + def __init__(self, variable, location: str, *, write_lock_factory=None): + """Wrap ``variable``, a thread-safe netCDF variable wrapper. + + Parameters + ---------- + variable : :class:`~iris.fileformats.netcdf._thread_safe_nc.VariableWrapper` + The wrapped netCDF variable. For character data this may be an + :class:`~iris.fileformats.netcdf._bytecoding_datasets.EncodedVariable`, + which reports a different shape, dimensions and dtype from the + file's own. This class reports whatever the wrapper reports. + location : str + The path or URL of the dataset this variable belongs to. + write_lock_factory : callable, optional + Returns the lock shared by every variable of one dataset. Called + by :meth:`write_handle`. A callable rather than a lock, so that a + read never makes one. + + """ + self._variable = variable + self._location = location + self._write_lock_factory = write_lock_factory + self._attributes = _NetCDFAttributes(variable) + + @property + def name(self) -> str: + """The variable's name within its dataset.""" + return self._variable.name + + @property + def location(self) -> str: + """The path or URL of the dataset holding this variable.""" + return self._location + + @property + def dimensions(self) -> tuple: + """The names of the dimensions this variable spans, in order.""" + return tuple(self._variable.dimensions) + + @property + def shape(self) -> tuple: + """The variable's shape.""" + return tuple(self._variable.shape) + + @property + def dtype(self): + """The variable's stored data type, or ``str`` for a VLEN string type.""" + return self._variable.dtype + + @property + def size(self) -> int: + """The total number of elements in the variable.""" + return self._variable.size + + @property + def fill_value(self) -> Any: + """The variable's ``_FillValue`` attribute, or ``None`` if it has none.""" + return self._attributes.get("_FillValue") + + @property + def chunking(self) -> tuple | None: + """The variable's storage chunk shape, or ``None`` when unchunked.""" + chunks = self._variable.chunking() + if chunks is None or chunks == _CONTIGUOUS: + return None + return tuple(chunks) + + @property + def attributes(self) -> MutableMapping: + """The variable's CF attributes.""" + return self._attributes + + def __getitem__(self, keys) -> np.ndarray: + """Return the indexed portion of the variable's data.""" + return self._variable[keys] + + def __setitem__(self, keys, values) -> None: + """Write ``values`` into the indexed portion of the variable.""" + self._variable[keys] = values + + def __repr__(self) -> str: + """Return a string representation.""" + return f"{self.__class__.__name__}({self.name!r}, {self.location!r})" + + # netCDF-only members below. + + @property + def variable(self): + """The backing thread-safe netCDF variable wrapper.""" + return self._variable + + @property + def unencoded_variable(self): + """The backing variable, from beneath any byte-encoding wrapper. + + An :class:`~iris.fileformats.netcdf._bytecoding_datasets.EncodedVariable` + hides the on-disk ``dtype`` and character dimension; callers needing + those want this rather than :attr:`variable`. An unwrapped variable is + returned unchanged. + + """ + if isinstance(self._variable, _bytecoding_datasets.EncodedVariable): + return self._variable._contained_instance + return self._variable + + @property + def is_variable_length(self) -> bool: + """Whether this is a netCDF variable-length (VLEN) type. + + Such a variable's total size cannot be known without reading it. + + """ + datatype = getattr(self._variable, "datatype", None) + return isinstance(datatype, _thread_safe_nc.VLType) + + @property + def is_emulated(self) -> bool: + """Whether an emulating object supplies this variable's data directly. + + ncdata, which bridges Iris and Xarray, passes a netCDF4 emulator + whose variables carry their own arrays instead of file storage. + + """ + return hasattr(self._variable, _EMULATED_DATA_ARRAY) + + @property + def emulated_data_array(self): + """The array an emulating variable carries instead of file storage.""" + if not self.is_emulated: + raise AttributeError(_EMULATED_DATA_ARRAY) + return getattr(self._variable, _EMULATED_DATA_ARRAY) + + @emulated_data_array.setter + def emulated_data_array(self, value) -> None: + if not self.is_emulated: + raise AttributeError(_EMULATED_DATA_ARRAY) + setattr(self._variable, _EMULATED_DATA_ARRAY, value) + + def deprecated_netcdf_member(self, name: str) -> Any: + """Return a member of the backing netCDF variable wrapper, with a warning.""" + # Fetch before warning: if getattr raises, no warning is issued. A + # failed hasattr() probe has not used the deprecated member. + value = getattr(self._variable, name) + warn_deprecated( + f"Reaching netCDF variable member {name!r} through a CFVariable is " + "deprecated and will be removed in a future release. Use the CF " + "dataset interface instead - iris.fileformats.cf.dataset - or, for " + "a netCDF-only need, cf_var.cf_data.variable." + ) + return value + + def write_handle(self) -> Any: + """Return a picklable object supporting ``__setitem__``, for Dask stores. + + It carries the file path and variable name rather than the open file, + and always encodes string data, whatever wrapper this variable arrived + in. + + """ + write_lock = None + if self._write_lock_factory is not None: + write_lock = self._write_lock_factory() + return _bytecoding_datasets.EncodedNetCDFWriteProxy( + self._location, self._variable, write_lock + ) + + +#: The formats that make CF loading slow, and that the user can convert away +#: from with "nccopy". +_LEGACY_FORMATS = ("NETCDF3_CLASSIC", "NETCDF3_64BIT") + + +class NetCDFDataset(CFDataset): + """A netCDF file, presented through the CF interface.""" + + def __init__( + self, + location, + mode: str = "r", + *, + netcdf_format=None, + warn_legacy_format: bool = False, + ): + """Open a netCDF file. + + Parameters + ---------- + location : str or :class:`pathlib.Path` + The file's path or URL. + mode : str, default="r" + The netCDF4 open mode. + netcdf_format : str, optional + The netCDF format to create, when writing. + warn_legacy_format : bool, default=False + Whether to warn that a netCDF3 file would load faster if + converted. Only loading asks for this; the loader is where the + user can act on it. + + """ + # Set first, so that __del__ and close() are safe if opening fails. + self._dataset: _thread_safe_nc.DatasetWrapper | None = None + self._owned = True + self._closed = False + self._variables: dict[str, NetCDFDatasetVariable] | None = None + self._attributes: _NetCDFAttributes | None = None + self._write_lock = None + + self._location = str(location) + self._mode = mode + + if mode == "r" and not _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ: + # The user has turned string decoding off for reads. Writing has + # no such switch: the saver always encodes. + dataset_class = _thread_safe_nc.DatasetWrapper + else: + dataset_class = _bytecoding_datasets.EncodedDataset + + # netCDF4.Dataset validates "format" against a fixed list and rejects + # None, so omit it entirely rather than passing the default through. + extra = {} if netcdf_format is None else {"format": netcdf_format} + self._dataset = dataset_class(location, mode=mode, **extra) + + if warn_legacy_format: + self._warn_if_legacy_format() + + # Turn off *any* automatic decoding by netCDF4 itself. Iris decodes + # byte data on its own terms, in _bytecoding_datasets. Inert on an + # EncodedDataset, which blocks the call; real on a DatasetWrapper. + self._dataset.set_auto_chartostring(False) + + def _warn_if_legacy_format(self) -> None: + """Warn that this file would load faster in netCDF4 format, if it would.""" + assert self._dataset is not None + if self._dataset.file_format in _LEGACY_FORMATS: + warnings.warn( + "Optimise CF-netCDF loading by converting data from NetCDF3 " + 'to NetCDF4 file format using the "nccopy" command.', + category=iris.warnings.IrisLoadWarning, + ) + + @classmethod + def from_existing( + cls, dataset, *, warn_legacy_format: bool = False + ) -> "NetCDFDataset": + """Wrap an already-open netCDF dataset, without taking ownership of it. + + ``dataset`` may be a thread-safe wrapper, a bare + :class:`netCDF4.Dataset`, or an emulator of one, as ncdata passes. + :meth:`close` will not release it, because whoever opened it is still + responsible for it. + + Parameters + ---------- + warn_legacy_format : bool, default=False + Whether to warn that a netCDF3 file would load faster if + converted. Only loading asks for this. + + """ + instance = cls.__new__(cls) + instance._owned = False + instance._closed = False + instance._variables = None + instance._attributes = None + instance._write_lock = None + # netCDF4 records no open mode, so this is stipulated, not observed. + # Only __repr__ reads it. + instance._mode = "r+" + + if not hasattr(dataset, "THREAD_SAFE_FLAG"): + # The wrappers forbid re-wrapping, so only wrap what is not one. + # Always encoded: DECODE_TO_STRINGS_ON_READ is not consulted here, + # because from_existing cannot tell a read from a write. + dataset = _bytecoding_datasets.EncodedDataset.from_existing(dataset) + instance._dataset = dataset + instance._dataset.set_auto_chartostring(False) + instance._location = str(dataset.filepath()) + + if warn_legacy_format: + instance._warn_if_legacy_format() + + return instance + + @property + def location(self) -> str: + """The file's path or URL.""" + return self._location + + @property + def mode(self) -> str: + """The mode the file was opened in.""" + return self._mode + + @property + def closed(self) -> bool: + """Whether the file has been released, by this object or by its owner. + + A borrowed dataset can be closed by its owner without this object + knowing, so the backing object is asked too. An emulator need not + implement ``isopen()``; one that does not is taken to be open. + + """ + if self._closed: + return True + isopen = getattr(self._dataset, "isopen", None) + if isopen is None: + return False + return not isopen() + + def _materialised_variables(self) -> dict[str, "NetCDFDatasetVariable"]: + """Return the wrapper mapping, building it from the file if need be.""" + if self._variables is None: + assert self._dataset is not None + self._variables = { + name: NetCDFDatasetVariable( + variable, + self._location, + write_lock_factory=self._write_lock_factory, + ) + for name, variable in self._dataset.variables.items() + } + return self._variables + + @property + def variables(self) -> Mapping: + """The file's variables, by name.""" + return self._materialised_variables() + + @property + def dimensions(self) -> Mapping: + """The file's dimension lengths, by name. + + An unlimited dimension reports the number of records written so far. + + """ + assert self._dataset is not None + return { + name: dimension.size for name, dimension in self._dataset.dimensions.items() + } + + @property + def attributes(self) -> MutableMapping: + """The file's global attributes.""" + if self._attributes is None: + self._attributes = _NetCDFAttributes(self._dataset) + return self._attributes + + def create_dimension(self, name: str, size: int | None) -> None: + """Declare a dimension of the given length, or unlimited for ``None``.""" + assert self._dataset is not None + self._dataset.createDimension(name, size) + + def create_variable( + self, name: str, dtype, dimensions=(), *, fill_value=None, **encoding + ) -> NetCDFDatasetVariable: + """Create and return a new variable. + + ``**encoding`` reaches :meth:`netCDF4.Dataset.createVariable` + unaltered: ``compression``, ``zlib``, ``complevel``, ``shuffle``, + ``chunksizes``, ``least_significant_digit`` and the rest. + + """ + assert self._dataset is not None + variable = self._dataset.createVariable( + name, dtype, tuple(dimensions), fill_value=fill_value, **encoding + ) + wrapped = NetCDFDatasetVariable( + variable, self._location, write_lock_factory=self._write_lock_factory + ) + # Register this wrapper, so that a later lookup by name returns it + # rather than a second wrapper with its own attribute cache. + self._materialised_variables()[name] = wrapped + return wrapped + + def sync(self) -> None: + """Flush buffered writes to the file.""" + assert self._dataset is not None + self._dataset.sync() + + def finalise(self) -> None: + """Do nothing: a netCDF file needs no completion step. + + Exists for stores that do, such as Zarr's consolidated metadata. + + """ + + def close(self) -> None: + """Close the file, if this object opened it.""" + if self._owned and self._dataset is not None and not self._closed: + self._dataset.close() + self._closed = True + + def __repr__(self) -> str: + """Return a string representation.""" + return f"{self.__class__.__name__}({self.location!r}, mode={self.mode!r})" + + # netCDF-only members below. + + @property + def dataset(self): + """The backing thread-safe netCDF dataset wrapper.""" + return self._dataset + + @property + def write_lock(self): + """The lock every worker writing to this file must hold. + + One per dataset. Made on first use, never at construction. + + """ + if self._write_lock is None: + self._write_lock = _dask_locks.get_worker_lock(self._location) + return self._write_lock + + def _write_lock_factory(self): + """Return :attr:`write_lock`, making it on the first call. + + Handed to every :class:`NetCDFDatasetVariable` so that each can find + the one shared lock at the moment it builds a write handle. + + """ + return self.write_lock diff --git a/lib/iris/fileformats/netcdf/loader.py b/lib/iris/fileformats/netcdf/loader.py index c12a3c6d44..8a1e942101 100644 --- a/lib/iris/fileformats/netcdf/loader.py +++ b/lib/iris/fileformats/netcdf/loader.py @@ -41,12 +41,8 @@ import iris.coord_systems import iris.coords import iris.fileformats.cf -from iris.fileformats.cf import CFDataVariable from iris.fileformats.netcdf import _bytecoding_datasets, _thread_safe_nc -from iris.fileformats.netcdf._bytecoding_datasets import ( - EncodedVariable, - VariableEncoder, -) +from iris.fileformats.netcdf._bytecoding_datasets import VariableEncoder from iris.fileformats.netcdf.saver import _CF_ATTRS import iris.io import iris.util @@ -216,10 +212,10 @@ def _get_actual_dtype(cf_var): # Figure out what the eventual data type will be after any scale/offset # transforms. dummy_data = np.zeros(1, dtype=cf_var.dtype) - if hasattr(cf_var, "scale_factor"): - dummy_data = cf_var.scale_factor * dummy_data - if hasattr(cf_var, "add_offset"): - dummy_data = cf_var.add_offset + dummy_data + if "scale_factor" in cf_var.attributes: + dummy_data = cf_var.attributes["scale_factor"] * dummy_data + if "add_offset" in cf_var.attributes: + dummy_data = cf_var.attributes["add_offset"] + dummy_data return dummy_data.dtype @@ -244,20 +240,20 @@ def _get_cf_var_data(cf_var): unnecessarily slow + wasteful of memory. """ - if hasattr(cf_var, "_data_array"): + if cf_var.cf_data.is_emulated: # The variable is not an actual netCDF4 file variable, but an emulating # object with an attached data array (either numpy or dask), which can be # returned immediately as-is. This is used as a hook to translate data to/from # netcdf data container objects in other packages, such as xarray. # See https://github.com/SciTools/iris/issues/4994 "Xarray bridge". - result = cf_var._data_array + result = cf_var.cf_data.emulated_data_array if result.dtype.kind == "S": # We must also perform any byte-to-string decoding since, in ncdata, the # emulating objects don't do this, and also don't support a # 'set_auto_chartostring(True)'. # Therefore, do here what an EncodedVariable.__getitem__ would do : .. # .. get details based on the file (type 'char') variable .. - encoder = VariableEncoder.from_var(cf_var.cf_data) + encoder = VariableEncoder.from_var(cf_var.cf_data.unencoded_variable) # .. convert byte array to strings. result = encoder.decode_bytes_to_stringarray(result) else: @@ -265,7 +261,7 @@ def _get_cf_var_data(cf_var): # netCDF arrays as the size of the array can only be known by reading the # data; see https://github.com/Unidata/netcdf-c/issues/1893. # Note: "Variable length" netCDF types have a datatype of `nc.VLType`. - if isinstance(getattr(cf_var, "datatype", None), _thread_safe_nc.VLType): + if cf_var.cf_data.is_variable_length: msg = ( f"NetCDF variable `{cf_var.cf_name}` is a variable length type of kind {cf_var.dtype} " "thus the total data size cannot be known in advance. This may affect the lazy loading " @@ -317,33 +313,33 @@ def _get_cf_var_data(cf_var): fill_value = "" else: fill_dtype = "S1" if cf_var.dtype is str else cf_var.dtype.str[1:] - fill_value = getattr( - cf_var.cf_data, - "_FillValue", - _thread_safe_nc.default_fillvals[fill_dtype], + fill_value = cf_var.attributes.get( + "_FillValue", _thread_safe_nc.default_fillvals[fill_dtype] ) # Switch type of proxy, based on type of variable. # It is done this way, instead of using an instance variable, because the # limited nature of the wrappers makes a stateful choice awkward, # e.g. especially, "variable.group()" is *not* the parent DatasetWrapper. - if isinstance(cf_var.cf_data, _bytecoding_datasets.EncodedVariable): + if isinstance( + cf_var.cf_data.variable, _bytecoding_datasets.EncodedVariable + ): proxy_class = _bytecoding_datasets.EncodedNetCDFDataProxy else: proxy_class = _thread_safe_nc.NetCDFDataProxy - proxy = proxy_class(cf_var.cf_data, dtype, cf_var.filename, fill_value) + proxy = proxy_class( + cf_var.cf_data.variable, dtype, cf_var.filename, fill_value + ) # Get the chunking specified for the variable : this is either a shape, or - # maybe the string "contiguous". + # None if the variable is unchunked. if CHUNK_CONTROL.mode is ChunkControl.Modes.AS_DASK: result = as_lazy_data(proxy, meta=proxy.dask_meta, chunks="auto") else: - chunks = cf_var.cf_data.chunking() + chunks = cf_var.cf_data.chunking if chunks is None: - # Occurs for non-version-4 netcdf - chunks = "contiguous" - # In the "contiguous" case, pass chunks=None to 'as_lazy_data'. - if chunks == "contiguous": + # Unchunked : either a non-version-4 file, or a contiguous + # version-4 variable. Neither offers a chunking to adopt. if ( CHUNK_CONTROL.mode is ChunkControl.Modes.FROM_FILE and isinstance(cf_var, iris.fileformats.cf.CFDataVariable) @@ -355,13 +351,16 @@ def _get_cf_var_data(cf_var): ) # Equivalent to chunks=None, but value required by chunking control chunks = list(cf_var.shape) + else: + # The chunk-control block below assigns into this. + chunks = list(chunks) # Modify the chunking in the context of an active chunking control. # N.B. settings specific to this named var override global ('*') ones. dim_chunks = CHUNK_CONTROL.var_dim_chunksizes.get( cf_var.cf_name ) or CHUNK_CONTROL.var_dim_chunksizes.get("*") - dims = cf_var.cf_data.dimensions + dims = cf_var.dimensions if CHUNK_CONTROL.mode is ChunkControl.Modes.FROM_FILE: dims_fixed = np.ones(len(dims), dtype=bool) elif not dim_chunks: @@ -429,7 +428,7 @@ def _load_cube_inner(engine, cf, cf_var, filename): """Create the cube associated with the CF-netCDF data variable.""" from iris.fileformats.netcdf.saver import Saver - if hasattr(cf_var, Saver._DATALESS_ATTRNAME): + if Saver._DATALESS_ATTRNAME in cf_var.attributes: # This data-variable represents a dataless cube. # The variable array content was never written (to take up no space). data = None @@ -659,12 +658,16 @@ def inner(cf_datavar): for name in constraint._names: expected = getattr(constraint, name) if name != "STASH" and expected != "none": - attr_name = "cf_name" if name == "var_name" else name - # Fetch property : N.B. CFVariable caches the property values - # The use of a default here is the only difference from the code in NameConstraint. - if not hasattr(cf_datavar, attr_name): + if name == "var_name": + # Iris's name for it; not a file attribute at all. + actual = cf_datavar.cf_name + elif name in cf_datavar.attributes: + actual = cf_datavar.attributes[name] + else: + # Unlike NameConstraint, a variable that does not + # carry the attribute is not a mismatch here: the + # cube may still acquire the name later in the load. continue - actual = getattr(cf_datavar, attr_name, "") if actual != expected: match_this_constraint = False break @@ -742,7 +745,7 @@ def load_cubes(file_sources, callback=None, constraints=None): mesh_name = None mesh = None mesh_coords, mesh_dim = [], None - mesh_name = getattr(cf_var, "mesh", None) + mesh_name = cf_var.attributes.get("mesh") if mesh_name is not None: try: mesh = meshes[mesh_name] diff --git a/lib/iris/fileformats/netcdf/saver.py b/lib/iris/fileformats/netcdf/saver.py index 232bfce66a..88a5fdda5e 100644 --- a/lib/iris/fileformats/netcdf/saver.py +++ b/lib/iris/fileformats/netcdf/saver.py @@ -63,13 +63,9 @@ import iris.exceptions import iris.fileformats.cf from iris.fileformats.netcdf import _bytecoding_datasets as bytecoding_datasets -from iris.fileformats.netcdf import _dask_locks -from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc from iris.fileformats.netcdf._attribute_handlers import ATTRIBUTE_HANDLERS -from iris.fileformats.netcdf._bytecoding_datasets import ( - EncodedVariable, - VariableEncoder, -) +from iris.fileformats.netcdf._bytecoding_datasets import VariableEncoder +from iris.fileformats.netcdf._dataset import NetCDFDataset, NetCDFDatasetVariable import iris.util import iris.warnings @@ -272,39 +268,6 @@ def coord(self, name): return result -def _bytes_if_ascii(string): - """Convert string to a byte string (str in py2k, bytes in py3k). - - Convert the given string to a byte string (str in py2k, bytes in py3k) - if the given string can be encoded to ascii, else maintain the type - of the inputted string. - - Note: passing objects without an `encode` method (such as None) will - be returned by the function unchanged. - - """ - if isinstance(string, str): - try: - return string.encode(encoding="ascii") - except (AttributeError, UnicodeEncodeError): - pass - return string - - -def _setncattr(variable, name, attribute): - """Put the given attribute on the given netCDF4 Data type. - - Put the given attribute on the given netCDF4 Data type, casting - attributes as we go to bytes rather than unicode. - - NOTE: variable needs to be a _thread_safe_nc._ThreadSafeWrapper subclass. - - """ - assert hasattr(variable, "THREAD_SAFE_FLAG") - attribute = _bytes_if_ascii(attribute) - return variable.setncattr(name, attribute) - - # NOTE : this matches :class:`iris.mesh.MeshXY.ELEMENTS`, # but in the preferred order for coord/connectivity variables in the file. MESH_ELEMENTS = ("node", "edge", "face") @@ -327,7 +290,10 @@ class VariableEmulator(typing.Protocol): shape: tuple[int, ...] -CFVariable = typing.Union[bytecoding_datasets.VariableWrapper, VariableEmulator] +# A saver variable is whatever the dataset hands back. For netCDF that is a +# NetCDFDatasetVariable, whether it wraps a real file variable or an +# emulating object - see VariableEmulator, above. +CFVariable = NetCDFDatasetVariable class Saver: @@ -430,7 +396,6 @@ def __init__(self, filename, netcdf_format, compute=True): self._to_open_dataset = hasattr(filename, "createVariable") if self._to_open_dataset: # We were passed a *dataset*, so we don't open (or close) one of our own. - self._dataset = filename if compute: msg = ( "Cannot save to a user-provided dataset with 'compute=True'. " @@ -439,15 +404,13 @@ def __init__(self, filename, netcdf_format, compute=True): ) raise ValueError(msg) - # Put it inside a _thread_safe_nc wrapper to ensure thread-safety. - # Except if it already is one, since they forbid "re-wrapping". - if not hasattr(self._dataset, "THREAD_SAFE_FLAG"): - self._dataset = bytecoding_datasets.EncodedDataset.from_existing( - self._dataset - ) + # from_existing() handles the thread-safety wrapping, including the + # case of an object that only emulates a dataset and carries no + # wrapper of its own. + self._dataset = NetCDFDataset.from_existing(filename) # In this case the dataset gives a filepath, not the other way around. - self.filepath = self._dataset.filepath() + self.filepath = self._dataset.location else: # Given a filepath string/path : create a dataset from that @@ -460,13 +423,13 @@ def __init__(self, filename, netcdf_format, compute=True): ) if self._is_nczarr: # NCZarr URLs contain a #mode= fragment; Path() strips it. - # Keep as a plain string and pass directly to DatasetWrapper. + # Keep as a plain string and pass directly to the dataset. self.filepath = str(filename) else: filepath = Path(filename) self.filepath = filepath.absolute() - self._dataset = bytecoding_datasets.EncodedDataset( - self.filepath, mode="w", format=netcdf_format + self._dataset = NetCDFDataset( + self.filepath, mode="w", netcdf_format=netcdf_format ) except RuntimeError: if self._is_nczarr: @@ -481,7 +444,10 @@ def __init__(self, filename, netcdf_format, compute=True): else: raise - self.file_write_lock = _dask_locks.get_worker_lock(self.filepath) + # One lock for the whole file, shared with every variable of it. A + # second get_worker_lock() call would hand back a different + # threading.Lock under the threaded scheduler, excluding nothing. + self.file_write_lock = self._dataset.write_lock def __enter__(self): return self @@ -684,8 +650,8 @@ def write( # N.B. _add_mesh cannot do this, as we want to put mesh variables # before data-variables in the file. if cf_mesh_name is not None: - _setncattr(cf_var_cube, "mesh", cf_mesh_name) - _setncattr(cf_var_cube, "location", cube.location) + cf_var_cube.attributes["mesh"] = cf_mesh_name + cf_var_cube.attributes["location"] = cube.location # Add coordinate variables. self._add_dim_coords(cube, cube_dimensions) @@ -734,7 +700,12 @@ def write( cf_patch = iris.site_configuration.get("cf_patch") if cf_patch is not None: # Perform a CF patch of the dataset. - cf_patch(profile, self._dataset, cf_var_cube) + # A reserved iris.site_configuration hook, documented since + # Iris 1.3 as receiving netCDF4 objects - so both arguments + # are unwrapped, not just the dataset. A CFVariable would + # raise on setncattr() and, worse, silently swallow + # 'variable.name = value': it defines no __setattr__. + cf_patch(profile, self._dataset.dataset, cf_var_cube.variable) else: msg = "cf_profile is available but no {} defined.".format("cf_patch") warnings.warn(msg, category=iris.warnings.IrisCfSaveWarning) @@ -792,10 +763,10 @@ def update_global_attributes(self, attributes=None, **kwargs): attributes = dict(attributes) for attr_name in sorted(attributes): - _setncattr(self._dataset, attr_name, attributes[attr_name]) + self._dataset.attributes[attr_name] = attributes[attr_name] for attr_name in sorted(kwargs): - _setncattr(self._dataset, attr_name, kwargs[attr_name]) + self._dataset.attributes[attr_name] = kwargs[attr_name] def _create_cf_dimensions(self, cube, dimension_names, unlimited_dimensions=None): """Create the CF-netCDF data dimensions. @@ -834,7 +805,7 @@ def _create_cf_dimensions(self, cube, dimension_names, unlimited_dimensions=None size = None else: size = self._existing_dim[dim_name] - self._dataset.createDimension(dim_name, size) + self._dataset.create_dimension(dim_name, size) def _add_mesh(self, cube_or_mesh, /, *, compression_kwargs=None): """Add the cube's mesh, and all related variables to the dataset. @@ -912,7 +883,7 @@ def _add_mesh(self, cube_or_mesh, /, *, compression_kwargs=None): # Record the coordinates (if any) on the mesh variable. if coord_names: coord_names = " ".join(coord_names) - _setncattr(cf_mesh_var, coords_file_attr, coord_names) + cf_mesh_var.attributes[coords_file_attr] = coord_names # Add all the connectivity variables. # pre-fetch the set + ignore "None"s, which are empty slots. @@ -930,7 +901,7 @@ def _add_mesh(self, cube_or_mesh, /, *, compression_kwargs=None): # See '_get_dim_names' for reason. last_dim = self._increment_name(last_dim) length = conn.shape[1 - conn.location_axis] - self._dataset.createDimension(last_dim, length) + self._dataset.create_dimension(last_dim, length) # Create variable. # NOTE: for connectivities *with missing points*, this will use a @@ -957,18 +928,18 @@ def _add_mesh(self, cube_or_mesh, /, *, compression_kwargs=None): ) # Add essential attributes to the Connectivity variable. cf_conn_var = self._dataset.variables[cf_conn_name] - _setncattr(cf_conn_var, "cf_role", cf_conn_attr_name) - _setncattr(cf_conn_var, "start_index", conn.start_index) + cf_conn_var.attributes["cf_role"] = cf_conn_attr_name + cf_conn_var.attributes["start_index"] = conn.start_index # Record the connectivity on the parent mesh var. - _setncattr(cf_mesh_var, cf_conn_attr_name, cf_conn_name) + cf_mesh_var.attributes[cf_conn_attr_name] = cf_conn_name # If the connectivity had the 'alternate' dimension order, add the # relevant dimension property if conn.location_axis == 1: loc_dim_attr = f"{loc_from}_dimension" # Should only get here once. - assert loc_dim_attr not in cf_mesh_var.ncattrs() - _setncattr(cf_mesh_var, loc_dim_attr, loc_dim_name) + assert loc_dim_attr not in cf_mesh_var.attributes + cf_mesh_var.attributes[loc_dim_attr] = loc_dim_name return cf_mesh_name @@ -1027,7 +998,7 @@ def _add_inner_related_vars( # Add CF-netCDF references to the primary data variable. if element_names: variable_names = " ".join(sorted(element_names)) - _setncattr(cf_var_cube, role_attribute_name, variable_names) + cf_var_cube.attributes[role_attribute_name] = variable_names def _add_aux_coords( self, cube, cf_var_cube, dimension_names, /, *, compression_kwargs=None @@ -1038,7 +1009,7 @@ def _add_aux_coords( ---------- cube : :class:`iris.cube.Cube` A :class:`iris.cube.Cube` to be saved to a netCDF file. - cf_var_cube : :class:`netcdf.netcdf_variable` + cf_var_cube : :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable` A cf variable cube representation. dimension_names : list Names associated with the dimensions of the cube. @@ -1075,7 +1046,7 @@ def _add_cell_measures(self, cube, cf_var_cube, dimension_names): ---------- cube : :class:`iris.cube.Cube` A :class:`iris.cube.Cube` to be saved to a netCDF file. - cf_var_cube : :class:`netcdf.netcdf_variable` + cf_var_cube : :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable` A cf variable cube representation. dimension_names : list Names associated with the dimensions of the cube. @@ -1096,7 +1067,7 @@ def _add_ancillary_variables( ---------- cube : :class:`iris.cube.Cube` A :class:`iris.cube.Cube` to be saved to a netCDF file. - cf_var_cube : :class:`netcdf.netcdf_variable` + cf_var_cube : :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable` A cf variable cube representation. dimension_names : list Names associated with the dimensions of the cube. @@ -1143,7 +1114,7 @@ def _add_aux_factories(self, cube, cf_var_cube, dimension_names): ---------- cube : :class:`iris.cube.Cube` A :class:`iris.cube.Cube` to be saved to a netCDF file. - cf_var_cube : :class:`netcdf.netcdf_variable` + cf_var_cube : :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable` CF variable cube representation. dimension_names : list Names associated with the dimensions of the cube. @@ -1183,10 +1154,10 @@ def _add_aux_factories(self, cube, cf_var_cube, dimension_names): ) std_name = factory_defn.std_name - if hasattr(cf_var, "formula_terms"): + if "formula_terms" in cf_var.attributes: if ( - cf_var.formula_terms != formula_terms - or cf_var.standard_name != std_name + cf_var.attributes["formula_terms"] != formula_terms + or cf_var.attributes["standard_name"] != std_name ): # TODO: We need to resolve this corner-case where # the dimensionless vertical coordinate containing @@ -1205,33 +1176,33 @@ def _add_aux_factories(self, cube, cf_var_cube, dimension_names): cube, dimension_names, primary_coord ) cf_var = self._dataset.variables[name] - _setncattr(cf_var, "standard_name", std_name) - _setncattr(cf_var, "axis", "Z") + cf_var.attributes["standard_name"] = std_name + cf_var.attributes["axis"] = "Z" # Update the formula terms. ft = formula_terms.split() ft = [name if t == cf_name else t for t in ft] - _setncattr(cf_var, "formula_terms", " ".join(ft)) + cf_var.attributes["formula_terms"] = " ".join(ft) # Update the cache. self._formula_terms_cache[key] = name # Update the associated cube variable. - coords = cf_var_cube.coordinates.split() + coords = cf_var_cube.attributes["coordinates"].split() coords = [name if c == cf_name else c for c in coords] - _setncattr(cf_var_cube, "coordinates", " ".join(coords)) + cf_var_cube.attributes["coordinates"] = " ".join(coords) else: - _setncattr(cf_var, "standard_name", std_name) - _setncattr(cf_var, "axis", "Z") - _setncattr(cf_var, "formula_terms", formula_terms) + cf_var.attributes["standard_name"] = std_name + cf_var.attributes["axis"] = "Z" + cf_var.attributes["formula_terms"] = formula_terms if FUTURE.derived_bounds: # ensure that the primary variable *bounds*, if any, obey the CF # encoding rule : the bounds variable of a parametric coordinate # must itself have a "formula_terms" attribute. # See : https://cfconventions.org/Data/cf-conventions/cf-conventions-1.12/cf-conventions.html#boundaries-and-formula-terms - bounds_varname = getattr(cf_var, "bounds", None) + bounds_varname = cf_var.attributes.get("bounds") cf_bounds_var = self._dataset.variables.get(bounds_varname, None) if ( cf_bounds_var is not None - and getattr(cf_bounds_var, "formula_terms", None) is None + and cf_bounds_var.attributes.get("formula_terms") is None ): # We need a bounds formula, and there is none already attached. # Construct and add one, mirroring the main formula. @@ -1241,7 +1212,13 @@ def boundsterm_varname(term_varname): result = term_varname # Follow links (if they exist) to find the bounds var. termvar = self._dataset.variables.get(term_varname) - boundsname = getattr(termvar, "bounds", None) + # An absent factory dependency has no variable name, + # so there is nothing to follow: keep the fallback. + boundsname = ( + None + if termvar is None + else termvar.attributes.get("bounds") + ) if boundsname in self._dataset.variables: result = boundsname return result @@ -1253,7 +1230,7 @@ def boundsterm_varname(term_varname): bounds_formula_terms = factory_defn.formula_terms_format.format( **boundsterm_varnames ) - _setncattr(cf_bounds_var, "formula_terms", bounds_formula_terms) + cf_bounds_var.attributes["formula_terms"] = bounds_formula_terms def _get_dim_names(self, cube_or_mesh): """Determine suitable CF-netCDF data dimension names. @@ -1517,7 +1494,7 @@ def _ensure_valid_dtype(self, values, src_name, src_object): if ( np.issubdtype(values.dtype, np.int64) or np.issubdtype(values.dtype, np.unsignedinteger) - ) and self._dataset.file_format in ( + ) and self._dataset.dataset.file_format in ( "NETCDF3_CLASSIC", "NETCDF3_64BIT", "NETCDF4_CLASSIC", @@ -1537,7 +1514,9 @@ def _ensure_valid_dtype(self, values, src_name, src_object): " its values cannot be safely cast to a supported" " integer type." ) - msg = msg.format(src_name, src_object, self._dataset.file_format) + msg = msg.format( + src_name, src_object, self._dataset.dataset.file_format + ) raise ValueError(msg) values = values.astype(np.int32) return values @@ -1589,11 +1568,11 @@ def _create_cf_bounds(self, coord, cf_var, cf_name, /, *, compression_kwargs=Non # Also avoid collision with variable names. # See '_get_dim_names' for reason. bounds_dimension_name = self._increment_name(bounds_dimension_name) - self._dataset.createDimension(bounds_dimension_name, n_bounds) + self._dataset.create_dimension(bounds_dimension_name, n_bounds) boundsvar_name = "{}_{}".format(cf_name, varname_extra) - _setncattr(cf_var, property_name, boundsvar_name) - cf_var_bounds = self._dataset.createVariable( + cf_var.attributes[property_name] = boundsvar_name + cf_var_bounds = self._dataset.create_variable( boundsvar_name, bounds.dtype.newbyteorder("="), cf_var.dimensions + (bounds_dimension_name,), @@ -1721,19 +1700,14 @@ def _create_mesh(self, mesh): cf_mesh_name = self._increment_name(cf_mesh_name) # Create the main variable - cf_mesh_var = self._dataset.createVariable( + cf_mesh_var = self._dataset.create_variable( cf_mesh_name, np.dtype(np.int32), - [], ) # Add the basic essential attributes - _setncattr(cf_mesh_var, "cf_role", "mesh_topology") - _setncattr( - cf_mesh_var, - "topology_dimension", - np.int32(mesh.topology_dimension), - ) + cf_mesh_var.attributes["cf_role"] = "mesh_topology" + cf_mesh_var.attributes["topology_dimension"] = np.int32(mesh.topology_dimension) # Add the usual names + units attributes self._set_cf_var_attributes(cf_mesh_var, mesh) @@ -1756,16 +1730,16 @@ def _set_cf_var_attributes(self, cf_var, element): # TODO: when we can break things, rationalise these to be the same. def add_units_attr(): if cf_units.as_unit(units_str).is_udunits(): - _setncattr(cf_var, "units", units_str) + cf_var.attributes["units"] = units_str def add_names_attrs(): standard_name = element.standard_name if standard_name is not None: - _setncattr(cf_var, "standard_name", standard_name) + cf_var.attributes["standard_name"] = standard_name long_name = element.long_name if long_name is not None: - _setncattr(cf_var, "long_name", long_name) + cf_var.attributes["long_name"] = long_name if isinstance(element, Cube): add_names_attrs() @@ -1776,7 +1750,7 @@ def add_names_attrs(): # Add the CF-netCDF calendar attribute. if element.units.calendar: - _setncattr(cf_var, "calendar", str(element.units.calendar)) + cf_var.attributes["calendar"] = str(element.units.calendar) # Take a copy so we can remove things element_attrs = element.attributes.copy() @@ -1788,7 +1762,7 @@ def add_names_attrs(): # *before* we can write to a character variable. if element.dtype.kind in "SU" and "_Encoding" in element_attrs: encoding = element_attrs.pop("_Encoding") - _setncattr(cf_var, "_Encoding", encoding) + cf_var.attributes["_Encoding"] = encoding if not isinstance(element, Cube): # Add any other custom coordinate attributes. @@ -1803,8 +1777,8 @@ def add_names_attrs(): value = str(value) # Don't clobber existing attributes. - if not hasattr(cf_var, name): - _setncattr(cf_var, name, value) + if name not in cf_var.attributes: + cf_var.attributes[name] = value def _create_generic_cf_array_var( self, @@ -1931,7 +1905,7 @@ def _create_generic_cf_array_var( # Also avoid collision with variable names. # See '_get_dim_names' for reason. string_dimension_name = self._increment_name(string_dimension_name) - self._dataset.createDimension( + self._dataset.create_dimension( string_dimension_name, string_dimension_depth ) @@ -1939,7 +1913,7 @@ def _create_generic_cf_array_var( element_dims.append(string_dimension_name) # Create the label coordinate variable. - cf_var = self._dataset.createVariable(cf_name, "|S1", element_dims) + cf_var = self._dataset.create_variable(cf_name, "|S1", element_dims) else: # A non-string variable. # ensure a valid datatype for the file format. @@ -1984,7 +1958,7 @@ def _create_generic_cf_array_var( cf_name = element_dims[0] # Create the CF-netCDF variable. - cf_var = self._dataset.createVariable( + cf_var = self._dataset.create_variable( cf_name, dtype, element_dims, @@ -1996,7 +1970,7 @@ def _create_generic_cf_array_var( if is_dimcoord: axis = iris.util.guess_coord_axis(element) if axis is not None and axis.lower() in SPATIO_TEMPORAL_AXES: - _setncattr(cf_var, "axis", axis.upper()) + cf_var.attributes["axis"] = axis.upper() # Create the associated CF-netCDF bounds variable, if any. self._create_cf_bounds( @@ -2012,7 +1986,7 @@ def _create_generic_cf_array_var( if packing_controls: # We must set packing attributes (if any), before assigning values. for key, value in packing_controls["attributes"]: - _setncattr(cf_var, key, value) + cf_var.attributes[key] = value self._lazy_stream_data(data=data, cf_var=cf_var) return cf_name @@ -2081,22 +2055,29 @@ def _add_grid_mapping_to_dataset(self, cs, extended_grid_mapping=False): ------- None """ - cf_var_grid = self._dataset.createVariable(cs.grid_mapping_name, np.int32) - _setncattr(cf_var_grid, "grid_mapping_name", cs.grid_mapping_name) + cf_var_grid = self._dataset.create_variable(cs.grid_mapping_name, np.int32) + cf_var_grid.attributes["grid_mapping_name"] = cs.grid_mapping_name + + # The sixty-three assignments below set CF grid-mapping parameters by + # Python attribute assignment. Unlike every other attribute the saver + # writes, they bypass the ASCII-to-bytes coercion, so moving them onto + # .attributes would change the file. See finding F8; until then they + # need the netCDF4 variable itself. + grid_variable = cf_var_grid.variable def add_ellipsoid(ellipsoid): - cf_var_grid.longitude_of_prime_meridian = ( + grid_variable.longitude_of_prime_meridian = ( ellipsoid.longitude_of_prime_meridian ) semi_major = ellipsoid.semi_major_axis semi_minor = ellipsoid.semi_minor_axis if semi_minor == semi_major: - cf_var_grid.earth_radius = semi_major + grid_variable.earth_radius = semi_major else: - cf_var_grid.semi_major_axis = semi_major - cf_var_grid.semi_minor_axis = semi_minor + grid_variable.semi_major_axis = semi_major + grid_variable.semi_minor_axis = semi_minor if ellipsoid.datum is not None: - cf_var_grid.horizontal_datum_name = ellipsoid.datum + grid_variable.horizontal_datum_name = ellipsoid.datum # latlon if isinstance(cs, iris.coord_systems.GeogCS): @@ -2106,19 +2087,23 @@ def add_ellipsoid(ellipsoid): elif isinstance(cs, iris.coord_systems.RotatedGeogCS): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.grid_north_pole_latitude = cs.grid_north_pole_latitude - cf_var_grid.grid_north_pole_longitude = cs.grid_north_pole_longitude - cf_var_grid.north_pole_grid_longitude = cs.north_pole_grid_longitude + grid_variable.grid_north_pole_latitude = cs.grid_north_pole_latitude + grid_variable.grid_north_pole_longitude = cs.grid_north_pole_longitude + grid_variable.north_pole_grid_longitude = cs.north_pole_grid_longitude # tmerc elif isinstance(cs, iris.coord_systems.TransverseMercator): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_central_meridian = cs.longitude_of_central_meridian - cf_var_grid.latitude_of_projection_origin = cs.latitude_of_projection_origin - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing - cf_var_grid.scale_factor_at_central_meridian = ( + grid_variable.longitude_of_central_meridian = ( + cs.longitude_of_central_meridian + ) + grid_variable.latitude_of_projection_origin = ( + cs.latitude_of_projection_origin + ) + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing + grid_variable.scale_factor_at_central_meridian = ( cs.scale_factor_at_central_meridian ) @@ -2126,16 +2111,16 @@ def add_ellipsoid(ellipsoid): elif isinstance(cs, iris.coord_systems.Mercator): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_projection_origin = ( + grid_variable.longitude_of_projection_origin = ( cs.longitude_of_projection_origin ) - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing # Only one of these should be set if cs.standard_parallel is not None: - cf_var_grid.standard_parallel = cs.standard_parallel + grid_variable.standard_parallel = cs.standard_parallel elif cs.scale_factor_at_projection_origin is not None: - cf_var_grid.scale_factor_at_projection_origin = ( + grid_variable.scale_factor_at_projection_origin = ( cs.scale_factor_at_projection_origin ) @@ -2143,38 +2128,38 @@ def add_ellipsoid(ellipsoid): elif isinstance(cs, iris.coord_systems.LambertConformal): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.standard_parallel = cs.secant_latitudes - cf_var_grid.latitude_of_projection_origin = cs.central_lat - cf_var_grid.longitude_of_central_meridian = cs.central_lon - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing + grid_variable.standard_parallel = cs.secant_latitudes + grid_variable.latitude_of_projection_origin = cs.central_lat + grid_variable.longitude_of_central_meridian = cs.central_lon + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing # polar stereo (have to do this before Stereographic because it subclasses it) elif isinstance(cs, iris.coord_systems.PolarStereographic): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.latitude_of_projection_origin = cs.central_lat - cf_var_grid.straight_vertical_longitude_from_pole = cs.central_lon - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing + grid_variable.latitude_of_projection_origin = cs.central_lat + grid_variable.straight_vertical_longitude_from_pole = cs.central_lon + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing # Only one of these should be set if cs.true_scale_lat is not None: - cf_var_grid.true_scale_lat = cs.true_scale_lat + grid_variable.true_scale_lat = cs.true_scale_lat elif cs.scale_factor_at_projection_origin is not None: - cf_var_grid.scale_factor_at_projection_origin = ( + grid_variable.scale_factor_at_projection_origin = ( cs.scale_factor_at_projection_origin ) else: - cf_var_grid.scale_factor_at_projection_origin = 1.0 + grid_variable.scale_factor_at_projection_origin = 1.0 # stereo elif isinstance(cs, iris.coord_systems.Stereographic): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_projection_origin = cs.central_lon - cf_var_grid.latitude_of_projection_origin = cs.central_lat - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing + grid_variable.longitude_of_projection_origin = cs.central_lon + grid_variable.latitude_of_projection_origin = cs.central_lat + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing # Only one of these should be set if cs.true_scale_lat is not None: msg = ( @@ -2183,11 +2168,11 @@ def add_ellipsoid(ellipsoid): ) raise ValueError(msg) elif cs.scale_factor_at_projection_origin is not None: - cf_var_grid.scale_factor_at_projection_origin = ( + grid_variable.scale_factor_at_projection_origin = ( cs.scale_factor_at_projection_origin ) else: - cf_var_grid.scale_factor_at_projection_origin = 1.0 + grid_variable.scale_factor_at_projection_origin = 1.0 # osgb (a specific tmerc) elif isinstance(cs, iris.coord_systems.OSGB): @@ -2200,47 +2185,57 @@ def add_ellipsoid(ellipsoid): elif isinstance(cs, iris.coord_systems.LambertAzimuthalEqualArea): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_projection_origin = ( + grid_variable.longitude_of_projection_origin = ( cs.longitude_of_projection_origin ) - cf_var_grid.latitude_of_projection_origin = cs.latitude_of_projection_origin - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing + grid_variable.latitude_of_projection_origin = ( + cs.latitude_of_projection_origin + ) + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing # albers conical equal area elif isinstance(cs, iris.coord_systems.AlbersEqualArea): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_central_meridian = cs.longitude_of_central_meridian - cf_var_grid.latitude_of_projection_origin = cs.latitude_of_projection_origin - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing - cf_var_grid.standard_parallel = cs.standard_parallels + grid_variable.longitude_of_central_meridian = ( + cs.longitude_of_central_meridian + ) + grid_variable.latitude_of_projection_origin = ( + cs.latitude_of_projection_origin + ) + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing + grid_variable.standard_parallel = cs.standard_parallels # vertical perspective elif isinstance(cs, iris.coord_systems.VerticalPerspective): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_projection_origin = ( + grid_variable.longitude_of_projection_origin = ( cs.longitude_of_projection_origin ) - cf_var_grid.latitude_of_projection_origin = cs.latitude_of_projection_origin - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing - cf_var_grid.perspective_point_height = cs.perspective_point_height + grid_variable.latitude_of_projection_origin = ( + cs.latitude_of_projection_origin + ) + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing + grid_variable.perspective_point_height = cs.perspective_point_height # geostationary elif isinstance(cs, iris.coord_systems.Geostationary): if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.longitude_of_projection_origin = ( + grid_variable.longitude_of_projection_origin = ( cs.longitude_of_projection_origin ) - cf_var_grid.latitude_of_projection_origin = cs.latitude_of_projection_origin - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing - cf_var_grid.perspective_point_height = cs.perspective_point_height - cf_var_grid.sweep_angle_axis = cs.sweep_angle_axis + grid_variable.latitude_of_projection_origin = ( + cs.latitude_of_projection_origin + ) + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing + grid_variable.perspective_point_height = cs.perspective_point_height + grid_variable.sweep_angle_axis = cs.sweep_angle_axis # oblique mercator (and rotated variant) # Use duck-typing over isinstance() - subclasses (i.e. @@ -2252,14 +2247,16 @@ def add_ellipsoid(ellipsoid): # all mention of RM. if cs.ellipsoid: add_ellipsoid(cs.ellipsoid) - cf_var_grid.azimuth_of_central_line = cs.azimuth_of_central_line - cf_var_grid.latitude_of_projection_origin = cs.latitude_of_projection_origin - cf_var_grid.longitude_of_projection_origin = ( + grid_variable.azimuth_of_central_line = cs.azimuth_of_central_line + grid_variable.latitude_of_projection_origin = ( + cs.latitude_of_projection_origin + ) + grid_variable.longitude_of_projection_origin = ( cs.longitude_of_projection_origin ) - cf_var_grid.false_easting = cs.false_easting - cf_var_grid.false_northing = cs.false_northing - cf_var_grid.scale_factor_at_projection_origin = ( + grid_variable.false_easting = cs.false_easting + grid_variable.false_northing = cs.false_northing + grid_variable.scale_factor_at_projection_origin = ( cs.scale_factor_at_projection_origin ) @@ -2274,7 +2271,7 @@ def add_ellipsoid(ellipsoid): # add WKT string if extended_grid_mapping: - cf_var_grid.crs_wkt = cs.as_cartopy_crs().to_wkt() + grid_variable.crs_wkt = cs.as_cartopy_crs().to_wkt() def _create_cf_grid_mapping(self, cube, cf_var_cube): """Create CF-netCDF grid mapping and associated CF-netCDF variable. @@ -2287,7 +2284,7 @@ def _create_cf_grid_mapping(self, cube, cf_var_cube): cube : :class:`iris.cube.Cube` or :class:`iris.cube.CubeList` A :class:`iris.cube.Cube`, :class:`iris.cube.CubeList` or list of cubes to be saved to a netCDF file. - cf_var_cube : :class:`netcdf.netcdf_variable` + cf_var_cube : :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable` A cf variable cube representation. Returns @@ -2396,7 +2393,7 @@ def _create_cf_grid_mapping(self, cube, cf_var_cube): grid_mapping = coord_systems[0].grid_mapping_name if grid_mapping: - _setncattr(cf_var_cube, "grid_mapping", grid_mapping) + cf_var_cube.attributes["grid_mapping"] = grid_mapping _DATALESS_ATTRNAME = "iris_dataless_cube" _DATALESS_DTYPE = np.dtype("u1") @@ -2556,17 +2553,17 @@ def _create_cf_data_variable( ) warnings.warn(msg, category=iris.warnings.IrisCfSaveWarning) - _setncattr(cf_var, attr_name, value) + cf_var.attributes[attr_name] = value # Add the 'dataless' marker if needed if is_dataless: - _setncattr(cf_var, self._DATALESS_ATTRNAME, "true") + cf_var.attributes[self._DATALESS_ATTRNAME] = "true" # Create the CF-netCDF data variable cell method attribute. cell_methods = self._create_cf_cell_methods(cube, dimension_names) if cell_methods: - _setncattr(cf_var, "cell_methods", cell_methods) + cf_var.attributes["cell_methods"] = cell_methods # Create the CF-netCDF grid mapping. self._create_cf_grid_mapping(cube, cf_var) @@ -2603,7 +2600,7 @@ def _increment_name(self, varname): def _lazy_stream_data( self, data: np.typing.ArrayLike, - cf_var: threadsafe_nc.VariableWrapper, + cf_var: NetCDFDatasetVariable, ) -> None: if hasattr(data, "shape") and data.shape == (1,) + cf_var.shape: # (Don't do this check for string data). @@ -2613,10 +2610,10 @@ def _lazy_stream_data( # contains just 1 row, so the cf_var is 1D. data = data.squeeze(axis=0) - if hasattr(cf_var, "_data_array"): - # The variable is not an actual netCDF4 file variable, but an emulating - # object with an attached data array (either numpy or dask), which should be - # copied immediately to the target. This is used as a hook to translate + if cf_var.is_emulated: + # The variable is not an actual file variable, but an emulating + # object with an attached data array (either numpy or dask), which should + # be copied immediately to the target. This is used as a hook to translate # data to/from netcdf data container objects in other packages, such as # xarray. # See https://github.com/SciTools/iris/issues/4994 "Xarray bridge". @@ -2626,11 +2623,11 @@ def _lazy_stream_data( # 'set_auto_chartostring(True)'. # Therefore, do here what an EncodedVariable.__setitem__ would do : .. # .. get details from the file (char) variable to be written .. - encoder = VariableEncoder.from_var(cf_var._contained_instance) + encoder = VariableEncoder.from_var(cf_var.unencoded_variable) # .. apply encoding to get the bytes to write. data = encoder.encode_strings_as_bytearray(data) - cf_var._data_array = data + cf_var.emulated_data_array = data else: doing_delayed_save = is_lazy_data(data) @@ -2642,27 +2639,23 @@ def _lazy_stream_data( return # save lazy data with a delayed operation. For now, we just record the - # necessary information -- a single, complete delayed action is constructed - # later by a call to delayed_completion(). + # necessary information -- a single, complete delayed action is + # constructed later by a call to delayed_completion(). def store( data: np.typing.ArrayLike, - cf_var: threadsafe_nc.VariableWrapper, + cf_var: NetCDFDatasetVariable, ) -> None: - # Create a data-writeable object that we can stream into, which - # encapsulates the file to be opened + variable to be written. - # Note: we do *not* support selectable string encoding for writes, - # so this never needs to be a _thread_safe_nc.NetCDFWriteProxy. - write_wrapper = bytecoding_datasets.EncodedNetCDFWriteProxy( - self.filepath, cf_var, self.file_write_lock - ) - # Add to the list of delayed writes, used in delayed_completion(). - self._delayed_writes.append((data, write_wrapper)) + # Ask the variable for something a worker can stream into + # after this file is closed. What that is is the backend's + # business: netCDF reopens the file, Zarr will hand back the + # array itself. + self._delayed_writes.append((data, cf_var.write_handle())) else: # Real data is always written directly, i.e. not via lazy save. def store( data: np.typing.ArrayLike, - cf_var: threadsafe_nc.VariableWrapper, + cf_var: NetCDFDatasetVariable, ) -> None: cf_var[:] = data # type: ignore[index] @@ -2705,7 +2698,7 @@ def complete(self) -> None: This requires that the Saver has closed the dataset (exited its context). """ - if self._dataset.isopen(): + if not self._dataset.closed: msg = ( "Cannot call Saver.complete() until its dataset is closed, " "i.e. the saver's context has exited." diff --git a/lib/iris/fileformats/netcdf/ugrid_load.py b/lib/iris/fileformats/netcdf/ugrid_load.py index 04a784b4f4..3d6b04fd9d 100644 --- a/lib/iris/fileformats/netcdf/ugrid_load.py +++ b/lib/iris/fileformats/netcdf/ugrid_load.py @@ -227,7 +227,7 @@ def _build_aux_coord(coord_var): climatological = False # TODO: use CF_ATTR_CLIMATOLOGY on re-integration, when no longer # 'experimental'. - attr_climatology = getattr(coord_var, "climatology", None) + attr_climatology = coord_var.attributes.get("climatology") if attr_climatology is not None: climatology_vars = coord_var.cf_group.climatology climatological = attr_climatology in climatology_vars @@ -319,11 +319,11 @@ def _build_mesh(cf, mesh_var): attr_units = get_attr_units(mesh_var, attributes) cf_role_message = None - if not hasattr(mesh_var, "cf_role"): + if "cf_role" not in mesh_var.attributes: cf_role_message = f"{mesh_var.cf_name} has no cf_role attribute." cf_role = "mesh_topology" else: - cf_role = getattr(mesh_var, "cf_role") + cf_role = mesh_var.attributes["cf_role"] if cf_role != "mesh_topology": cf_role_message = f"{mesh_var.cf_name} has an inappropriate cf_role: {cf_role}." if cf_role_message: @@ -333,17 +333,17 @@ def _build_mesh(cf, mesh_var): category=_WarnComboCfDefaulting, ) - if hasattr(mesh_var, "volume_node_connectivity"): + if "volume_node_connectivity" in mesh_var.attributes: topology_dimension = 3 - elif hasattr(mesh_var, "face_node_connectivity"): + elif "face_node_connectivity" in mesh_var.attributes: topology_dimension = 2 - elif hasattr(mesh_var, "edge_node_connectivity"): + elif "edge_node_connectivity" in mesh_var.attributes: topology_dimension = 1 else: # Nodes only. We aren't sure yet whether this is a valid option. topology_dimension = 0 - if not hasattr(mesh_var, "topology_dimension"): + if "topology_dimension" not in mesh_var.attributes: msg = ( f"MeshXY variable {mesh_var.cf_name} has no 'topology_dimension'" f" : *Assuming* topology_dimension={topology_dimension}" @@ -351,7 +351,7 @@ def _build_mesh(cf, mesh_var): ) warnings.warn(msg, category=_WarnComboCfDefaulting) else: - quoted_topology_dimension = mesh_var.topology_dimension + quoted_topology_dimension = mesh_var.attributes["topology_dimension"] if quoted_topology_dimension != topology_dimension: msg = ( f"*Assuming* 'topology_dimension'={topology_dimension}" @@ -367,8 +367,8 @@ def _build_mesh(cf, mesh_var): ) node_dimension = None - edge_dimension = getattr(mesh_var, "edge_dimension", None) - face_dimension = getattr(mesh_var, "face_dimension", None) + edge_dimension = mesh_var.attributes.get("edge_dimension") + face_dimension = mesh_var.attributes.get("face_dimension") node_coord_args = [] edge_coord_args = [] @@ -377,12 +377,12 @@ def _build_mesh(cf, mesh_var): coord_and_axis = _build_aux_coord(coord_var) coord = coord_and_axis[0] - if coord.var_name in mesh_var.node_coordinates.split(): + if coord.var_name in mesh_var.attributes["node_coordinates"].split(): node_coord_args.append(coord_and_axis) node_dimension = coord_var.dimensions[0] - elif coord.var_name in getattr(mesh_var, "edge_coordinates", "").split(): + elif coord.var_name in mesh_var.attributes.get("edge_coordinates", "").split(): edge_coord_args.append(coord_and_axis) - elif coord.var_name in getattr(mesh_var, "face_coordinates", "").split(): + elif coord.var_name in mesh_var.attributes.get("face_coordinates", "").split(): face_coord_args.append(coord_and_axis) # TODO: support volume_coordinates. else: @@ -403,7 +403,7 @@ def _build_mesh(cf, mesh_var): connectivity, first_dim_name = _build_connectivity( connectivity_var, element_dims ) - assert connectivity.var_name == getattr(mesh_var, connectivity.cf_role) + assert connectivity.var_name == mesh_var.attributes[connectivity.cf_role] connectivity_args.append(connectivity) # If the mesh_var has not supplied the dimension name, it is safe to @@ -453,12 +453,13 @@ def _build_mesh_coords(mesh, cf_var): "edge": mesh.edge_dimension, "face": mesh.face_dimension, } - location = getattr(cf_var, "location", "") + location = cf_var.attributes.get("location", "") if location is None or location not in element_dimensions: # We should probably issue warnings and recover, but that is too much # work. Raising a more intelligible error is easy to do though. msg = ( - f"mesh data variable {cf_var.name!r} has an invalid location={location!r}." + f"mesh data variable {cf_var.cf_name!r} has an invalid " + f"location={location!r}." ) raise ValueError(msg) mesh_dim_name = element_dimensions.get(location) @@ -469,10 +470,10 @@ def _build_mesh_coords(mesh, cf_var): mesh_dim = cf_var.dimensions.index(mesh_dim_name) else: msg = ( - f"mesh data variable {cf_var.name!r} does not have the " + f"mesh data variable {cf_var.cf_name!r} does not have the " f"{location} mesh dimension {mesh_dim_name!r}, in its dimensions." ) raise ValueError(msg) - mesh_coords = mesh.to_MeshCoords(location=cf_var.location) + mesh_coords = mesh.to_MeshCoords(location=location) return mesh_coords, mesh_dim diff --git a/lib/iris/tests/integration/netcdf/derived_bounds/test_bounds_files.py b/lib/iris/tests/integration/netcdf/derived_bounds/test_bounds_files.py index c7d3564fd2..396bd35ce5 100644 --- a/lib/iris/tests/integration/netcdf/derived_bounds/test_bounds_files.py +++ b/lib/iris/tests/integration/netcdf/derived_bounds/test_bounds_files.py @@ -18,6 +18,10 @@ import iris from iris import FUTURE, sample_data_path +from iris.aux_factory import HybridHeightFactory +from iris.coords import AuxCoord, DimCoord +from iris.cube import Cube +from iris.fileformats.netcdf._thread_safe_nc import DatasetWrapper from iris.tests._shared_utils import assert_CDL from iris.tests.stock.netcdf import ncgen_from_cdl @@ -220,3 +224,57 @@ def test_save_primary_cf_style( assert_CDL( request=request, netcdf_filename=nc_filepath, reference_filename=cdl_filepath ) + + +@pytest.fixture +def cube_with_absent_orography(): + """Return a hybrid-height cube whose factory has no orography. + + :class:`~iris.aux_factory.HybridHeightFactory` documents ``orography`` as + optional, so a saved ``formula_terms`` can name a dependency that was + never written as a variable. + """ + cube = Cube(np.zeros(3, dtype="f8"), standard_name="air_temperature", units="K") + delta = DimCoord( + np.arange(3.0), + long_name="level_height", + var_name="level_height", + units="m", + bounds=np.array([[0.0, 1.0], [1.0, 2.0], [2.0, 3.0]]), + ) + sigma = AuxCoord( + np.linspace(1.0, 0.0, 3), + long_name="sigma", + var_name="sigma", + units="1", + bounds=np.array([[1.0, 0.75], [0.75, 0.25], [0.25, 0.0]]), + ) + cube.add_dim_coord(delta, 0) + cube.add_aux_coord(sigma, 0) + cube.add_aux_factory(HybridHeightFactory(delta=delta, sigma=sigma, orography=None)) + return cube + + +def test_save_absent_factory_dependency(cube_with_absent_orography, tmp_ncdir): + """Save a factory with an absent dependency, with derived bounds enabled. + + The bounds ``formula_terms`` is built by following each term variable's + ``bounds`` link, and an absent dependency has no term variable to follow. + Asserts on the file rather than on the save completing: the terms this + writes are the pre-existing behaviour being preserved, not an improvement + on it, and a fix that silently dropped ``orog`` would also not raise. + """ + nc_filepath = tmp_ncdir / "test_save_absent_orography.nc" + + with FUTURE.context(derived_bounds=True): + iris.save(cube_with_absent_orography, nc_filepath) + + dataset = DatasetWrapper(nc_filepath) + try: + formula_terms = dataset.variables["level_height"].formula_terms + bounds_formula_terms = dataset.variables["level_height_bnds"].formula_terms + finally: + dataset.close() + + assert formula_terms == "a: level_height b: sigma orog: None" + assert bounds_formula_terms == "a: level_height_bnds b: sigma_bnds orog: None" diff --git a/lib/iris/tests/integration/netcdf/test__dask_locks.py b/lib/iris/tests/integration/netcdf/test__dask_locks.py index 1aee902195..7d56adb974 100644 --- a/lib/iris/tests/integration/netcdf/test__dask_locks.py +++ b/lib/iris/tests/integration/netcdf/test__dask_locks.py @@ -15,8 +15,11 @@ import dask import dask.config import distributed +import numpy as np import pytest +import iris +from iris.cube import Cube from iris.fileformats.netcdf._dask_locks import ( DaskSchedulerTypeError, dask_scheduler_is_distributed, @@ -109,3 +112,31 @@ def test_get_worker_lock(dask_scheduler): else: # low-level object doesn't have a readily available class for isinstance assert all(hasattr(result, att) for att in ("acquire", "release", "locked")) + + +def test_load_does_not_need_a_worker_lock(tmp_path): + """A load must not demand a lock only the saver needs. + + ``get_worker_lock`` refuses the process scheduler, because the *saver* + cannot use it - see this module's own ``test_get_worker_lock``. Loading + has no such limitation, so no read path may ask for the lock: doing so + made ``iris.load`` fail under ``scheduler="processes"``, reporting that + the scheduler "is not supported by the Iris netcdf saver" during a load. + + The process scheduler is really configured rather than simulated, because + that is the configuration a user hits - and configuring it is the whole + trigger: the failure was at lock construction during the load, before any + compute. No worker process is in fact spawned, because this file is below + ``loader._LAZYVAR_MIN_BYTES`` and so loads eagerly, which is what keeps + the test quick and deterministic under pytest. + """ + path = tmp_path / "process_scheduler.nc" + with iris.FUTURE.context(save_split_attrs=True): + iris.save(Cube(np.arange(6.0).reshape(2, 3), long_name="x"), path) + + with dask.config.set(scheduler="processes"): + loaded = iris.load(path) + data = loaded[0].data + + assert len(loaded) == 1 + np.testing.assert_array_equal(data, np.arange(6.0).reshape(2, 3)) diff --git a/lib/iris/tests/integration/netcdf/test_attributes.py b/lib/iris/tests/integration/netcdf/test_attributes.py index b24538079f..d715972b7a 100644 --- a/lib/iris/tests/integration/netcdf/test_attributes.py +++ b/lib/iris/tests/integration/netcdf/test_attributes.py @@ -7,11 +7,16 @@ from contextlib import contextmanager from cf_units import Unit +import numpy as np import pytest import iris +import iris.coord_systems +from iris.coords import DimCoord from iris.cube import Cube, CubeList from iris.fileformats.netcdf import CF_CONVENTIONS_VERSION +from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc +from iris.fileformats.netcdf._dataset import NetCDFDatasetVariable from iris.tests import _shared_utils @@ -101,6 +106,177 @@ def test_patching_conventions_attribute(self, tmp_path, mocker): ) +class TestCfPatch: + """What the ``cf_patch`` hook is handed is public API. + + ``cf_patch`` is a reserved :data:`iris.site_configuration` key, documented + since Iris 1.3 as receiving netCDF4 objects. Handing it a + :class:`~iris.fileformats.netcdf._dataset.NetCDFDatasetVariable` instead + breaks it two ways: ``setncattr`` raises, and a plain + ``variable.name = value`` is discarded in silence, since a CF variable + defines no ``__setattr__``. + + """ + + def test_cf_patch_receives_a_netcdf4_variable(self, tmp_path): + received = [] + + def cf_profile(cube): + return "a profile" + + def cf_patch(profile, dataset, variable): + received.append(variable) + # Write through the netCDF4 API the hook is promised. + variable.setncattr("patched_by_cf_patch", "cf_patch was here") + + # A real function, not a Mock: a Mock accepts setncattr() whatever it + # is given, and records a call that never reached a file. + orig_site_config = iris.site_configuration.copy() + iris.site_configuration["cf_profile"] = cf_profile + iris.site_configuration["cf_patch"] = cf_patch + nc_path = tmp_path / "cf_patch.nc" + try: + cube = Cube([1.0], standard_name="air_temperature", units="K") + iris.save(cube, nc_path) + finally: + iris.site_configuration = orig_site_config + + (variable,) = received + assert not isinstance(variable, NetCDFDatasetVariable) + + # The read-back is the point. Asserting that the hook ran, or that it + # did not raise, would miss an attribute going quietly nowhere. + result = iris.load_cube(nc_path) + assert result.attributes["patched_by_cf_patch"] == "cf_patch was here" + + +class TestAttributesNamedLikeNetcdf4Members: + """A coordinate attribute may be named after a netCDF4 Python member. + + ``shape``, ``size`` and the rest below are members of a netCDF4 Variable + object as well as plausible attribute names. The saver's "don't clobber" + check used to ask ``hasattr``, so every one of them was silently dropped + on the way to the file. It asks the CF attribute mapping now, which knows + the difference between an attribute of the data and a member of the + object holding it, and the six reach the file like any other. + + """ + + NAMES = ("shape", "size", "name", "dtype", "dimensions", "mask") + + def test_attributes_named_after_netcdf4_members_are_saved(self, tmp_path): + coord = DimCoord( + np.arange(3.0), + standard_name="longitude", + units="degrees", + attributes={name: f"value of {name}" for name in self.NAMES}, + ) + cube = Cube(np.arange(3.0), standard_name="air_temperature", units="K") + cube.add_dim_coord(coord, 0) + + nc_path = tmp_path / "netcdf4_member_names.nc" + with iris.FUTURE.context(save_split_attrs=True): + iris.save(cube, nc_path) + + # Reading back is the only way to see this: the dropped attributes + # were dropped in silence, with no warning and no error. + result = iris.load_cube(nc_path).coord("longitude") + assert {name: result.attributes.get(name) for name in self.NAMES} == { + name: f"value of {name}" for name in self.NAMES + } + + +class TestGridMappingAttributes: + """Grid-mapping parameters are written by plain attribute assignment. + + Every other attribute the saver writes goes through the CF attribute + mapping. These do not - they are set straight onto the netCDF4 variable, + because doing otherwise would change their type in the file (finding F8). + That makes a mistake in one of them silent: assigning to the wrong object, + or misspelling the object, leaves a stray Python attribute and no + attribute in the file at all. + + These two projections are here because they carry the eight parameter + assignments no other test reads back. + + """ + + @staticmethod + def saved_grid_mapping(coord_system, tmp_path): + """Save a cube on this coord system; return the grid-mapping attributes.""" + cube = Cube( + np.zeros((2, 3), dtype=np.float32), + standard_name="air_temperature", + units="K", + ) + for index, axis in enumerate("yx"): + cube.add_dim_coord( + DimCoord( + np.arange(cube.shape[index], dtype=np.float64), + standard_name=f"projection_{axis}_coordinate", + units="m", + coord_system=coord_system, + ), + index, + ) + nc_path = tmp_path / f"{coord_system.grid_mapping_name}.nc" + with iris.FUTURE.context(save_split_attrs=True): + iris.save(cube, nc_path) + + # Read the file itself rather than a loaded cube: what is being + # pinned is that these values reach the variable, not what the + # loader is able to reconstruct from them. + dataset = threadsafe_nc.DatasetWrapper(nc_path) + try: + variable = dataset.variables[coord_system.grid_mapping_name] + return {name: variable.getncattr(name) for name in variable.ncattrs()} + finally: + dataset.close() + + def test_mercator_scale_factor(self, tmp_path): + # Mercator takes a scale factor *or* a standard parallel. Only the + # standard-parallel branch is read back anywhere else. + coord_system = iris.coord_systems.Mercator( + longitude_of_projection_origin=90.0, + scale_factor_at_projection_origin=0.9, + ) + attributes = self.saved_grid_mapping(coord_system, tmp_path) + assert attributes["scale_factor_at_projection_origin"] == 0.9 + assert "standard_parallel" not in attributes + + @pytest.mark.parametrize( + ("scale_kwargs", "expected_scale"), + [ + ({"true_scale_lat": 71.0}, {"true_scale_lat": 71.0}), + ( + {"scale_factor_at_projection_origin": 0.9}, + {"scale_factor_at_projection_origin": 0.9}, + ), + # Neither given: the saver writes a scale factor of 1.0 rather + # than leaving the projection unscaled. + ({}, {"scale_factor_at_projection_origin": 1.0}), + ], + ids=["true_scale_lat", "scale_factor", "neither"], + ) + def test_polar_stereographic(self, scale_kwargs, expected_scale, tmp_path): + coord_system = iris.coord_systems.PolarStereographic( + central_lat=90.0, + central_lon=-150.0, + false_easting=13.0, + false_northing=17.0, + **scale_kwargs, + ) + attributes = self.saved_grid_mapping(coord_system, tmp_path) + expected = { + "latitude_of_projection_origin": 90.0, + "straight_vertical_longitude_from_pole": -150.0, + "false_easting": 13.0, + "false_northing": 17.0, + **expected_scale, + } + assert {name: attributes.get(name) for name in expected} == expected + + class TestStandardName: def test_standard_name_roundtrip(self, tmp_path): standard_name = "air_temperature detection_minimum" diff --git a/lib/iris/tests/test_netcdf.py b/lib/iris/tests/test_netcdf.py index 2605343c4d..1a1bfdf438 100644 --- a/lib/iris/tests/test_netcdf.py +++ b/lib/iris/tests/test_netcdf.py @@ -24,6 +24,7 @@ from iris.tests import _shared_utils import iris.tests.stock as stock from iris.tests.stock.netcdf import ncgen_from_cdl +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble import iris.util from iris.warnings import IrisCfSaveWarning @@ -466,16 +467,13 @@ def test_units(self, request): class TestNetCDFCRS: @pytest.fixture(autouse=True) def _setup(self): - class Var: - pass - - self.grid = Var() + self.grid = CFVariableDouble() def test_lat_lon_major_minor(self): major = 63781370 minor = 63567523 - self.grid.semi_major_axis = major - self.grid.semi_minor_axis = minor + self.grid.attributes["semi_major_axis"] = major + self.grid.attributes["semi_minor_axis"] = minor # NB 'build_coordinate_system' has an extra (unused) 'engine' arg, just # so that it has the same signature as other coord builder routines. engine = None @@ -484,7 +482,7 @@ def test_lat_lon_major_minor(self): def test_lat_lon_earth_radius(self): earth_radius = 63700000 - self.grid.earth_radius = earth_radius + self.grid.attributes["earth_radius"] = earth_radius # NB 'build_coordinate_system' has an extra (unused) 'engine' arg, just # so that it has the same signature as other coord builder routines. engine = None diff --git a/lib/iris/tests/unit/fileformats/cf/dataset/__init__.py b/lib/iris/tests/unit/fileformats/cf/dataset/__init__.py new file mode 100644 index 0000000000..f50199ed98 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/cf/dataset/__init__.py @@ -0,0 +1,5 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :mod:`iris.fileformats.cf.dataset`.""" diff --git a/lib/iris/tests/unit/fileformats/cf/dataset/contract.py b/lib/iris/tests/unit/fileformats/cf/dataset/contract.py new file mode 100644 index 0000000000..1eab7536de --- /dev/null +++ b/lib/iris/tests/unit/fileformats/cf/dataset/contract.py @@ -0,0 +1,264 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""One test body, run against every :class:`CFDataset` implementation. + +Subclass :class:`CFDatasetContract` in a module named for the implementation, +supply the three fixtures it declares, and every test here runs against it. +The class deliberately has no ``Test`` prefix, so pytest does not collect it +where it is written, only where it is subclassed. + +Spec section 6: the interface is only worth having if both sides of it agree, +and agreement is cheapest to check by asking the same questions twice. + +Deliberately not exercised here: :meth:`CFDatasetVariable.deprecated_netcdf_member`. +It is an explicitly netCDF4-specific compatibility escape hatch, so asserting +either its presence or its absence in a shared body would itself be +backend-specific - the omission is intentional, not a gap. +""" + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable + +#: The canonical dataset every implementation is checked against. +CONTRACT_DIMENSIONS = {"y": 3, "x": 2} +CONTRACT_DATA = np.arange(6, dtype="f8").reshape(3, 2) +CONTRACT_VARIABLE_ATTRS = {"units": "K", "long_name": "surface temperature"} +CONTRACT_GLOBALS = {"Conventions": "CF-1.7", "title": "contract"} + + +def populate(dataset): + """Fill an empty writable dataset with the canonical contents. + + Uses only the CFDataset API, so this doubles as the write half of the + contract: an implementation that cannot build this cannot be used to save. + """ + for name, size in CONTRACT_DIMENSIONS.items(): + dataset.create_dimension(name, size) + for name, value in CONTRACT_GLOBALS.items(): + dataset.attributes[name] = value + + variable = dataset.create_variable( + "air_temperature", np.dtype("f8"), tuple(CONTRACT_DIMENSIONS), fill_value=-1.0 + ) + for name, value in CONTRACT_VARIABLE_ATTRS.items(): + variable.attributes[name] = value + variable[:] = CONTRACT_DATA + + # A dimensionless variable: what a grid mapping is, and the one shape a + # CFDatasetVariable must handle without a leading dimension. + scalar = dataset.create_variable("grid", np.dtype("i4")) + scalar.attributes["grid_mapping_name"] = "latitude_longitude" + scalar[()] = 0 + return dataset + + +class CFDatasetContract: + """What every CFDataset implementation must do, whatever it stores into.""" + + @pytest.fixture + def readable(self): + """Return an open read-mode dataset holding what populate() writes.""" + raise NotImplementedError("supply a 'readable' fixture") + + @pytest.fixture + def writable(self): + """Return an open, empty, write-mode dataset.""" + raise NotImplementedError("supply a 'writable' fixture") + + @pytest.fixture + def reopen(self): + """Return a callable giving a fresh read-mode dataset over 'writable'.""" + raise NotImplementedError("supply a 'reopen' fixture") + + # -- The dataset itself ------------------------------------------------ + + def test_is_a_cf_dataset(self, readable): + assert isinstance(readable, CFDataset) + + def test_location_is_a_non_empty_string(self, readable): + assert isinstance(readable.location, str) + assert readable.location + + def test_mode_is_a_read_mode(self, readable): + assert readable.mode in ("r", "r+", "a") + + def test_starts_open(self, readable): + assert readable.closed is False + + def test_close_then_closed(self, writable): + writable.close() + assert writable.closed is True + + def test_close_is_idempotent(self, writable): + writable.close() + writable.close() + assert writable.closed is True + + def test_context_manager_returns_self_and_closes(self, writable): + with writable as entered: + assert entered is writable + assert writable.closed is True + + def test_dimensions(self, readable): + assert dict(readable.dimensions) == CONTRACT_DIMENSIONS + + def test_global_attributes(self, readable): + assert dict(readable.attributes) == CONTRACT_GLOBALS + + def test_variable_names(self, readable): + assert sorted(readable.variables) == ["air_temperature", "grid"] + + # -- A variable -------------------------------------------------------- + + @pytest.fixture + def variable(self, readable): + return readable.variables["air_temperature"] + + @pytest.fixture + def scalar(self, readable): + return readable.variables["grid"] + + def test_variable_is_a_cf_dataset_variable(self, variable): + assert isinstance(variable, CFDatasetVariable) + + def test_variable_name(self, variable): + assert variable.name == "air_temperature" + + def test_variable_location_matches_its_dataset(self, variable, readable): + assert variable.location == readable.location + + def test_variable_dimensions(self, variable): + assert variable.dimensions == tuple(CONTRACT_DIMENSIONS) + + def test_variable_shape(self, variable): + assert variable.shape == tuple(CONTRACT_DIMENSIONS.values()) + + def test_variable_dtype(self, variable): + assert variable.dtype == np.dtype("f8") + + def test_variable_size(self, variable): + assert variable.size == CONTRACT_DATA.size + + def test_variable_ndim(self, variable): + assert variable.ndim == CONTRACT_DATA.ndim + + def test_variable_len(self, variable): + assert len(variable) == CONTRACT_DATA.shape[0] + + def test_variable_data(self, variable): + np.testing.assert_array_equal(variable[:], CONTRACT_DATA) + + def test_variable_data_indexed(self, variable): + np.testing.assert_array_equal(variable[1], CONTRACT_DATA[1]) + + def test_variable_attributes(self, variable): + for name, value in CONTRACT_VARIABLE_ATTRS.items(): + assert variable.attributes[name] == value + + def test_variable_attributes_omit_unset_names(self, variable): + assert "nonesuch" not in variable.attributes + with pytest.raises(KeyError): + variable.attributes["nonesuch"] + + def test_variable_fill_value(self, variable): + assert variable.fill_value == -1.0 + + def test_variable_chunking_is_none_or_a_shape(self, variable): + chunking = variable.chunking + # np.integer alongside int: netCDF4 answers plain ints, but a Zarr + # chunk shape is plausibly numpy.int64, and both are equally a shape. + assert chunking is None or ( + isinstance(chunking, tuple) + and len(chunking) == len(variable.shape) + and all(isinstance(size, int | np.integer) for size in chunking) + ) + + def test_scalar_variable(self, scalar): + assert scalar.dimensions == () + assert scalar.shape == () + with pytest.raises(TypeError, match="unsized"): + len(scalar) + + # -- Writing ----------------------------------------------------------- + + def test_populate_then_read_back(self, writable, reopen): + populate(writable) + writable.close() + with reopen() as reader: + np.testing.assert_array_equal( + reader.variables["air_temperature"][:], CONTRACT_DATA + ) + assert dict(reader.dimensions) == CONTRACT_DIMENSIONS + assert dict(reader.attributes) == CONTRACT_GLOBALS + + def test_created_variable_is_visible_on_the_dataset(self, writable): + writable.create_dimension("x", 2) + created = writable.create_variable("thing", np.dtype("f4"), ("x",)) + assert writable.variables["thing"].name == created.name + + def test_create_variable_without_dimensions(self, writable): + # Finding F4: saver.py:2080 supplies only a name and a dtype. + variable = writable.create_variable("grid", np.dtype("i4")) + assert variable.dimensions == () + + def test_unlimited_dimension_is_supported_or_refused(self, writable): + # The one member whose docstring sanctions two answers: a store with + # no unlimited concept - Zarr - refuses rather than inventing one. + # Either answer conforms; what does not conform is a third, such as + # quietly creating a fixed-length dimension instead. + try: + writable.create_dimension("t", None) + except NotImplementedError: + return + assert "t" in writable.dimensions + + def test_setitem_then_getitem(self, writable): + writable.create_dimension("x", 3) + variable = writable.create_variable("thing", np.dtype("f4"), ("x",)) + variable[:] = [1.0, 2.0, 3.0] + np.testing.assert_array_equal(variable[:], [1.0, 2.0, 3.0]) + + def test_attribute_write_through(self, writable, reopen): + populate(writable) + writable.variables["air_temperature"].attributes["comment"] = "added" + writable.close() + with reopen() as reader: + assert reader.variables["air_temperature"].attributes["comment"] == "added" + + def test_write_handle_works_after_close(self, writable, reopen): + # Not asserted here: that `handle` itself survives a pickle round + # trip. CFDatasetVariable.write_handle's docstring requires it, + # because it is what a Dask worker receives as a da.store target - + # but the handle carries a backend-injected lock + # (NetCDFDataset.write_lock) whose type depends on the active Dask + # scheduler, and under the default threaded scheduler that lock is + # an unpicklable threading.Lock which is, in practice, never + # pickled: _dask_locks.get_worker_lock only returns a picklable + # dask.distributed.Lock under the distributed scheduler, which is + # also the only scheduler that ever ships a handle to another + # process. A round-trip assertion here would wrongly fail a correct + # backend under the default scheduler. What an implementation + # actually owes is that the handle itself captures no unpicklable + # state of its own - no open file, store handle or live connection - + # which is what this test approximates by proving the handle still + # works once the dataset that created it has closed. + populate(writable) + handle = writable.variables["air_temperature"].write_handle() + writable.close() + + payload = CONTRACT_DATA * 10 + handle[:] = payload + with reopen() as reader: + np.testing.assert_array_equal( + reader.variables["air_temperature"][:], payload + ) + + def test_sync_and_finalise_are_callable(self, writable): + populate(writable) + assert writable.sync() is None + assert writable.finalise() is None + assert writable.closed is False diff --git a/lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py b/lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py new file mode 100644 index 0000000000..022ed1e35a --- /dev/null +++ b/lib/iris/tests/unit/fileformats/cf/dataset/test_CFDataset.py @@ -0,0 +1,151 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.cf.dataset.CFDataset` and friends.""" + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDataset, CFDatasetVariable + + +class MinimalVariable(CFDatasetVariable): + """The smallest thing that satisfies CFDatasetVariable.""" + + name = "air_temperature" + location = "" + dimensions = ("time", "lat") + shape = (3, 4) + dtype = np.dtype("f4") + size = 12 + fill_value = None + chunking = None + attributes: dict = {} + + def __getitem__(self, keys): + return np.zeros(self.shape, dtype=self.dtype)[keys] + + def __setitem__(self, keys, values): + raise NotImplementedError + + def write_handle(self): + return self + + +class MinimalDataset(CFDataset): + """The smallest thing that satisfies CFDataset.""" + + location = "" + mode = "r" + closed = False + variables: dict = {} + dimensions: dict = {} + attributes: dict = {} + + def __init__(self): + self.closes = 0 + + def create_dimension(self, name, size): + raise NotImplementedError + + def create_variable(self, name, dtype, dimensions=(), *, fill_value=None, **kw): + raise NotImplementedError + + def sync(self): + pass + + def finalise(self): + pass + + def close(self): + self.closes += 1 + + +class TestAbstractness: + def test_variable_cannot_be_instantiated(self): + with pytest.raises(TypeError, match="abstract"): + CFDatasetVariable() + + def test_dataset_cannot_be_instantiated(self): + with pytest.raises(TypeError, match="abstract"): + CFDataset() + + @pytest.mark.parametrize( + "name", + [ + "name", + "location", + "dimensions", + "shape", + "dtype", + "size", + "fill_value", + "chunking", + "attributes", + "ndim", + "__getitem__", + "__setitem__", + "write_handle", + ], + ) + def test_variable_declares_member(self, name): + assert name in CFDatasetVariable.__abstractmethods__ or hasattr( + CFDatasetVariable, name + ) + + @pytest.mark.parametrize( + "name", + [ + "location", + "mode", + "closed", + "variables", + "dimensions", + "attributes", + "create_dimension", + "create_variable", + "sync", + "finalise", + "close", + ], + ) + def test_dataset_declares_member(self, name): + assert name in CFDataset.__abstractmethods__ or hasattr(CFDataset, name) + + +class TestConcreteDefaults: + def test_ndim(self): + assert MinimalVariable().ndim == 2 + + def test_len_is_the_leading_dimension(self): + assert len(MinimalVariable()) == 3 + + def test_len_of_a_scalar_matches_netcdf4(self): + # netCDF4.Variable raises TypeError, and CFVariable.__len__ forwards + # to it today, so anything catching that keeps working. + class Scalar(MinimalVariable): + shape = () + + with pytest.raises(TypeError, match="unsized"): + len(Scalar()) + + def test_deprecated_netcdf_member_raises_attribute_error(self): + # A store with no backing netCDF4 object - Zarr, say - has no fallback + # to offer, so the compatibility route simply does not apply. + with pytest.raises(AttributeError, match="getncattr"): + MinimalVariable().deprecated_netcdf_member("getncattr") + + def test_context_manager_closes(self): + dataset = MinimalDataset() + with dataset as entered: + assert entered is dataset + assert dataset.closes == 0 + assert dataset.closes == 1 + + def test_context_manager_closes_on_exception(self): + dataset = MinimalDataset() + with pytest.raises(ValueError, match="boom"): + with dataset: + raise ValueError("boom") + assert dataset.closes == 1 diff --git a/lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py b/lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py new file mode 100644 index 0000000000..17ac882be5 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/cf/dataset/test_TrackedAttributes.py @@ -0,0 +1,123 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.cf.dataset.TrackedAttributes`.""" + +import pytest + +from iris.fileformats.cf.dataset import TrackedAttributes + +# The set CFVariable seeds tracking with: attributes netCDF4 handles itself, +# and which are therefore "used" before anyone reads them. +IGNORED = ("_FillValue", "add_offset", "missing_value", "scale_factor") + + +@pytest.fixture +def source(): + return {"units": "K", "standard_name": "air_temperature", "_FillValue": -999} + + +@pytest.fixture +def tracked(source): + return TrackedAttributes(source, ignored=IGNORED) + + +class TestWhatRecordsARead: + def test_starts_with_only_the_present_ignored_names(self, tracked): + # "scale_factor" is ignored but absent, so it is not seeded. + assert tracked.read == frozenset(["_FillValue"]) + assert tracked.unread == frozenset(["units", "standard_name"]) + + def test_getitem_records(self, tracked): + assert tracked["units"] == "K" + assert tracked.read == frozenset(["_FillValue", "units"]) + assert tracked.unread == frozenset(["standard_name"]) + + def test_get_records(self, tracked): + assert tracked.get("units") == "K" + assert "units" in tracked.read + + def test_contains_records_a_hit(self, tracked): + # hasattr(cf_var, name) goes through __getattr__ today and marks the + # attribute used; the mapping form has to do the same. + assert "units" in tracked + assert "units" in tracked.read + + def test_contains_does_not_record_a_miss(self, tracked): + assert "nonesuch" not in tracked + assert tracked.read == frozenset(["_FillValue"]) + + def test_getitem_of_a_missing_key_raises_before_recording(self, tracked): + with pytest.raises(KeyError, match="nonesuch"): + tracked["nonesuch"] + assert tracked.read == frozenset(["_FillValue"]) + + def test_get_of_a_missing_key_records_nothing(self, tracked): + assert tracked.get("nonesuch") is None + assert tracked.read == frozenset(["_FillValue"]) + + +class TestWhatDoesNotRecordARead: + def test_untracked_getitem(self, tracked): + assert tracked.untracked["units"] == "K" + assert tracked.read == frozenset(["_FillValue"]) + + def test_untracked_contains(self, tracked): + # The form helpers.py needs for its deliberately-unmarked flag probe. + assert "units" in tracked.untracked + assert tracked.read == frozenset(["_FillValue"]) + + def test_untracked_is_read_only(self, tracked): + with pytest.raises(TypeError): + tracked.untracked["units"] = "m" + + def test_iteration(self, tracked): + assert sorted(tracked) == ["_FillValue", "standard_name", "units"] + assert tracked.read == frozenset(["_FillValue"]) + + def test_keys_values_items(self, tracked): + assert sorted(tracked.keys()) == ["_FillValue", "standard_name", "units"] + assert sorted(tracked.values(), key=str) == [-999, "K", "air_temperature"] + assert dict(tracked.items())["units"] == "K" + assert tracked.read == frozenset(["_FillValue"]) + + def test_len(self, tracked): + assert len(tracked) == 3 + assert tracked.read == frozenset(["_FillValue"]) + + +class TestMutation: + def test_setitem_writes_through_and_does_not_record(self, tracked, source): + tracked["comment"] = "hello" + assert source["comment"] == "hello" + assert tracked.read == frozenset(["_FillValue"]) + + def test_delitem_writes_through_and_forgets_the_read(self, tracked, source): + _ = tracked["units"] + del tracked["units"] + assert "units" not in source + assert tracked.read == frozenset(["_FillValue"]) + + +class TestReset: + def test_reset_returns_to_the_present_ignored_names(self, tracked): + _ = tracked["units"] + _ = tracked["standard_name"] + assert tracked.read == frozenset(["_FillValue", "units", "standard_name"]) + tracked.reset() + assert tracked.read == frozenset(["_FillValue"]) + + def test_reset_sees_attributes_added_since_construction(self, tracked): + tracked["scale_factor"] = 2.0 + tracked.reset() + assert tracked.read == frozenset(["_FillValue", "scale_factor"]) + + +class TestEmptySource: + def test_a_variable_with_no_attributes_is_not_an_error(self): + tracked = TrackedAttributes({}, ignored=IGNORED) + assert tracked.read == frozenset() + assert tracked.unread == frozenset() + assert len(tracked) == 0 + assert "units" not in tracked diff --git a/lib/iris/tests/unit/fileformats/cf/identify_mixins.py b/lib/iris/tests/unit/fileformats/cf/identify_mixins.py index a99699a054..d2c54f56bf 100644 --- a/lib/iris/tests/unit/fileformats/cf/identify_mixins.py +++ b/lib/iris/tests/unit/fileformats/cf/identify_mixins.py @@ -15,6 +15,7 @@ import pytest from iris.fileformats.cf import CFVariable +from iris.fileformats.cf._variables import _NCZARR_SCALAR_DIMENSION import iris.warnings @@ -34,6 +35,11 @@ def ncattrs(self): if not attr.startswith("_") and attr not in self.ATTRS_NOT_RETURN ] + @property + def attributes(self): + """The CF attributes, as a CFDatasetVariable presents them.""" + return {name: getattr(self, name) for name in self.ncattrs()} + class _NetCDFVarWithDimensions(_NetCDFVar): """Stub NetCDF variable with dimensions for spans() tests.""" @@ -97,6 +103,27 @@ def test_non_spanning(self): cf_target = self._make_cf_var("target_var", ("x", "y")) assert not cf_source.spans(cf_target) + def test_nczarr_scalar_dimension_spans(self): + """An NCZarr scalar source variable always spans the target. + + NCZarr spells a zero-dimensional variable as one dimension named + _scalar_, so this is the same case as test_empty_dimensions_spans + arriving from a different writer. + """ + cf_source = self._make_cf_var("source_var", (_NCZARR_SCALAR_DIMENSION,)) + cf_target = self._make_cf_var("target_var", ("x", "y")) + assert cf_source.spans(cf_target) + + def test_nczarr_scalar_is_not_a_dimension_name_to_match_on(self): + """_scalar_ marks a scalar; it is not a dimension two variables share. + + A source that is not scalar has to justify itself dimension by + dimension, whether or not _scalar_ is among its names. + """ + cf_source = self._make_cf_var("source_var", ("a", "b", "extra")) + cf_target = self._make_cf_var("target_var", (_NCZARR_SCALAR_DIMENSION,)) + assert not cf_source.spans(cf_target) + class IdentifyByAttributeMixin(ABC): """Parent class for CF variable identify() tests. diff --git a/lib/iris/tests/unit/fileformats/cf/test_CFCoordinateVariable.py b/lib/iris/tests/unit/fileformats/cf/test_CFCoordinateVariable.py index 8747689298..3acdf3a5ee 100644 --- a/lib/iris/tests/unit/fileformats/cf/test_CFCoordinateVariable.py +++ b/lib/iris/tests/unit/fileformats/cf/test_CFCoordinateVariable.py @@ -26,6 +26,10 @@ def __init__(self, name, dimensions, data, dtype=float): def ncattrs(self): return [] + @property + def attributes(self): + return {} + def __getitem__(self, key): if self._data.ndim == 0: return self._data diff --git a/lib/iris/tests/unit/fileformats/cf/test_CFReader.py b/lib/iris/tests/unit/fileformats/cf/test_CFReader.py index 37af98641c..41e933e49a 100644 --- a/lib/iris/tests/unit/fileformats/cf/test_CFReader.py +++ b/lib/iris/tests/unit/fileformats/cf/test_CFReader.py @@ -28,6 +28,16 @@ import iris.warnings +# CFVariable.attributes, and now identify() too, are built from a variable's +# ncattrs()/getncattr() - see CFVariable.__init__ and CFReader._translate. +# `netcdf_variable` below wires ncattrs()/getncattr() to a presence-keyed +# dict of the CF-attribute keywords the caller passed here - fixed at this +# call. A test that mutates or deletes one of these named attributes on the +# returned mock afterwards - as a few in this file do, to simulate a variable +# missing an attribute - changes only the raw Mock attribute, which neither +# identify() nor `.attributes` ever reads again. To vary what a variable's +# attributes are, pass the value to a fresh `netcdf_variable(...)` call +# instead - see e.g. `test_derived_bounds_promotes_reference_terms` below. def netcdf_variable( mocker, name, @@ -41,6 +51,12 @@ def netcdf_variable( grid_mapping=None, cell_measures=None, standard_name=None, + long_name=None, + mesh=None, + cf_role=None, + node_coordinates=None, + face_coordinates=None, + face_node_connectivity=None, ): """Return a mock NetCDF4 variable.""" ndim = 0 @@ -49,18 +65,17 @@ def netcdf_variable( ndim = len(dimensions) else: dimensions = [] + # Arbitrary but real: NetCDFDatasetVariable.shape/.ndim read straight + # through to these, once this mock is wrapped rather than read directly. + shape = tuple(1 for _ in range(ndim)) + size = 1 ugrid_identities = ( CFUGridAuxiliaryCoordinateVariable.cf_identities + CFUGridConnectivityVariable.cf_identities + [CFUGridMeshVariable.cf_identity] ) - ncvar = mocker.Mock( - name=name, - dimensions=dimensions, - ncattrs=mocker.Mock(return_value=[]), - ndim=ndim, - dtype=dtype, + members = dict( ancillary_variables=ancillary_variables, coordinates=coordinates, bounds=bounds, @@ -69,8 +84,38 @@ def netcdf_variable( grid_mapping=grid_mapping, cell_measures=cell_measures, standard_name=standard_name, + long_name=long_name, + cf_role=cf_role, **{name: None for name in ugrid_identities}, ) + # Each of these is one of the UGRID identities the sweep above defaults + # to None; the explicit keywords are what let a test set them. + members["mesh"] = mesh + members["node_coordinates"] = node_coordinates + members["face_coordinates"] = face_coordinates + members["face_node_connectivity"] = face_node_connectivity + # A None member stood for "no such attribute" when identify() read these + # with getattr(..., None). Now that it reads a mapping, absent means + # absent - so a None must not be listed. + attributes = {key: value for key, value in members.items() if value is not None} + ncvar = mocker.Mock( + name=name, + dimensions=dimensions, + ndim=ndim, + shape=shape, + size=size, + dtype=dtype, + ncattrs=mocker.Mock(return_value=list(attributes)), + getncattr=mocker.Mock(side_effect=attributes.__getitem__), + # A few tests in this file build a CFVariable straight from this + # mock, bypassing NetCDFDatasetVariable entirely; CFVariable.__init__ + # reads `.attributes` directly, so it has to be present here too - + # harmless for the usual path, where NetCDFDatasetVariable computes + # its own `.attributes` from ncattrs()/getncattr() and never looks at + # this one. + attributes=attributes, + **members, + ) return ncvar @@ -171,21 +216,21 @@ def test_create_formula_terms(self, mocker): group = cf_group.data_variables assert len(group) == 1 assert list(group.keys()) == ["temp"] - assert group["temp"].cf_data is self.temp + assert group["temp"].cf_data.variable is self.temp # Check there are three coordinates. group = cf_group.coordinates assert len(group) == 3 coordinates = ["height", "lat", "lon"] assert set(group.keys()) == set(coordinates) for name in coordinates: - assert group[name].cf_data is getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) # Check there are three auxiliary coordinates. group = cf_group.auxiliary_coordinates assert len(group) == 3 aux_coordinates = ["delta", "sigma", "orography"] assert set(group.keys()) == set(aux_coordinates) for name in aux_coordinates: - assert group[name].cf_data is getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) # Check all the auxiliary coordinates are formula terms. formula_terms = cf_group.formula_terms assert set(group.items()) == set(formula_terms.items()) @@ -195,7 +240,7 @@ def test_create_formula_terms(self, mocker): bounds = ["height_bnds", "delta_bnds", "sigma_bnds"] assert set(group.keys()) == set(bounds) for name in bounds: - assert group[name].cf_data == getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) class Test_build_cf_groups__formula_terms: @@ -285,19 +330,19 @@ def test_associate_formula_terms_with_data_variable(self, mocker): coordinates = ["height", "lat", "lon"] assert set(group.keys()) == set(coordinates) for name in coordinates: - assert group[name].cf_data is getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) # Check the height coordinate is bounded. group = group["height"].cf_group assert len(group.bounds) == 1 assert "height_bnds" in group.bounds - assert group["height_bnds"].cf_data is self.height_bnds + assert group["height_bnds"].cf_data.variable is self.height_bnds # Check there are five auxiliary coordinates. group = temp_cf_group.auxiliary_coordinates assert len(group) == 5 aux_coordinates = ["delta", "sigma", "orography", "x", "y"] assert set(group.keys()) == set(aux_coordinates) for name in aux_coordinates: - assert group[name].cf_data is getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) # Check all the auxiliary coordinates are formula terms. formula_terms = cf_group.formula_terms assert set(formula_terms.items()).issubset(list(group.items())) @@ -309,7 +354,9 @@ def test_associate_formula_terms_with_data_variable(self, mocker): aux_coord_group = group[name].cf_group assert len(aux_coord_group.bounds) == 1 assert name_bnds in aux_coord_group.bounds - assert aux_coord_group[name_bnds].cf_data is getattr(self, name_bnds) + assert aux_coord_group[name_bnds].cf_data.variable is getattr( + self, name_bnds + ) def test_promote_reference(self): cf_group = CFReader("dummy").cf_group @@ -326,7 +373,7 @@ def test_promote_reference(self): coordinates = ("lat", "lon") assert set(group.keys()) == set(coordinates) for name in coordinates: - assert group[name].cf_data == getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) def test_formula_terms_ignore(self): self.orography.dimensions = ["lat", "wibble"] @@ -334,7 +381,7 @@ def test_formula_terms_ignore(self): cf_group = CFReader("dummy").cf_group group = cf_group.promoted assert list(group.keys()) == ["orography"] - assert group["orography"].cf_data == self.orography + assert group["orography"].cf_data.variable is self.orography def test_auxiliary_ignore(self): self.x.dimensions = ["lat", "wibble"] @@ -344,11 +391,16 @@ def test_auxiliary_ignore(self): group = cf_group.promoted assert set(group.keys()) == set(promoted) for name in promoted: - assert group[name].cf_data == getattr(self, name) + assert group[name].cf_data.variable is getattr(self, name) - def test_promoted_auxiliary_ignore(self): + def test_promoted_auxiliary_ignore(self, mocker): self.variables["wibble"] = self.wibble - self.orography.coordinates = "wibble" + # "coordinates" must be set at construction - see the module-level + # comment on netcdf_variable(). + self.orography = netcdf_variable( + mocker, "orography", "lat lon", np.float64, coordinates="wibble" + ) + self.variables["orography"] = self.orography with pytest.warns(match="Ignoring variable wibble") as warns: cf_group = CFReader("dummy").cf_group.promoted @@ -356,7 +408,7 @@ def test_promoted_auxiliary_ignore(self): promoted = ["wibble", "orography"] assert set(cf_group.keys()) == set(promoted) for name in promoted: - assert cf_group[name].cf_data == getattr(self, name) + assert cf_group[name].cf_data.variable is getattr(self, name) # we should have got 2 warnings assert len(warns.list) == 2 @@ -365,25 +417,41 @@ class Test_build_cf_groups__ugrid: @pytest.fixture(autouse=True) def _setup_class(self, mocker): # Replicating syntax from test_CFReader.Test_build_cf_groups__formula_terms. - self.mesh = netcdf_variable(mocker, "mesh", "", int) + # Mesh-recognition attributes are passed at construction, not set + # afterwards: identify() now reads through .attributes, which is + # fixed when netcdf_variable() builds ncattrs()/getncattr() - see the + # module-level comment above. + self.mesh = netcdf_variable( + mocker, + "mesh", + "", + int, + cf_role="mesh_topology", + node_coordinates="node_x node_y", + face_coordinates="face_x face_y", + face_node_connectivity="face_nodes", + ) self.node_x = netcdf_variable(mocker, "node_x", "node", float) self.node_y = netcdf_variable(mocker, "node_y", "node", float) self.face_x = netcdf_variable(mocker, "face_x", "face", float) self.face_y = netcdf_variable(mocker, "face_y", "face", float) - self.face_nodes = netcdf_variable(mocker, "face_nodes", "face vertex", int) + self.face_nodes = netcdf_variable( + mocker, + "face_nodes", + "face vertex", + int, + cf_role="face_node_connectivity", + ) self.levels = netcdf_variable(mocker, "levels", "levels", int) self.data = netcdf_variable( - mocker, "data", "levels face", float, coordinates="face_x face_y" + mocker, + "data", + "levels face", + float, + coordinates="face_x face_y", + mesh="mesh", ) - # Add necessary attributes for mesh recognition. - self.mesh.cf_role = "mesh_topology" - self.mesh.node_coordinates = "node_x node_y" - self.mesh.face_coordinates = "face_x face_y" - self.mesh.face_node_connectivity = "face_nodes" - self.face_nodes.cf_role = "face_node_connectivity" - self.data.mesh = "mesh" - self.variables = dict( mesh=self.mesh, node_x=self.node_x, @@ -520,6 +588,8 @@ def _setup(self, mocker): variables=self.variables, ncattrs=mocker.Mock(return_value=[]), filepath=mocker.Mock(return_value="in-memory.nc"), + set_auto_chartostring=mocker.Mock(), + isopen=mocker.Mock(return_value=True), ) self.encoded_ds = mocker.patch( "iris.fileformats.netcdf._bytecoding_datasets.EncodedDataset", @@ -543,7 +613,7 @@ def test_init_uses_dataset_wrapper_when_string_decode_disabled(self, mocker): False, ) wrapper_ds = mocker.patch( - "iris.fileformats.cf._reader._thread_safe_nc.DatasetWrapper", + "iris.fileformats.netcdf._thread_safe_nc.DatasetWrapper", return_value=self.dataset, ) @@ -566,8 +636,20 @@ def test_init_warns_for_netcdf3_when_requested(self): with pytest.warns(iris.warnings.IrisLoadWarning, match="Optimise CF-netCDF"): CFReader("dummy.nc", warn=True) + def test_init_warns_for_netcdf3_when_requested_for_a_borrowed_dataset(self): + # The owned-path equivalent above pins CFReader passing warn= through + # to NetCDFDataset; this pins the same wiring on the from_existing + # branch, which is a separate call site in __init__. + self.dataset.file_format = "NETCDF3_CLASSIC" + + with pytest.warns(iris.warnings.IrisLoadWarning, match="Optimise CF-netCDF"): + CFReader(self.dataset, warn=True) + def test_init_with_no_meshes_trims_ugrid_variable_types(self, mocker): - self.dataset.variables = {"a": object(), "b": mocker.Mock(mesh=None)} + self.dataset.variables = { + "a": netcdf_variable(mocker, "a", "x", np.float64), + "b": netcdf_variable(mocker, "b", "x", np.float64, mesh="my_mesh"), + } reader = CFReader("dummy.nc") @@ -575,9 +657,10 @@ def test_init_with_no_meshes_trims_ugrid_variable_types(self, mocker): mesh_free = mocker.Mock( file_format="NetCDF4", - variables={}, + variables={"a": netcdf_variable(mocker, "a", "x", np.float64)}, ncattrs=mocker.Mock(return_value=[]), filepath=mocker.Mock(return_value="in-memory.nc"), + set_auto_chartostring=mocker.Mock(), ) self.encoded_ds.return_value = mesh_free reader = CFReader("dummy.nc") @@ -686,7 +769,12 @@ def test_derived_bounds_term(self, mocker, future_context): assert isinstance(cf_group["term"], CFAuxiliaryCoordinateVariable) def test_derived_bounds_skips_when_term_missing(self, mocker, future_context): - self.root_bnds.formula_terms = "a: missing_term" + # "formula_terms" must be set at construction - see the module-level + # comment on netcdf_variable(). + self.root_bnds = netcdf_variable( + mocker, "z_bnds", "z bnds", np.float64, formula_terms="a: missing_term" + ) + self.variables["z_bnds"] = self.root_bnds self._patch_encoded_dataset(mocker) with future_context: @@ -699,7 +787,6 @@ def test_promotes_non_formula_root_bounds_to_data(self, mocker): self.root_bnds = netcdf_variable(mocker, "z_bnds", "_scalar_", np.float64) # With valid formula terms, the variable would instead be recorded # correctly as a bounds variable. - del self.root_bnds.formula_terms self.variables["z_bnds"] = self.root_bnds self._patch_encoded_dataset(mocker) @@ -792,19 +879,20 @@ def test_formula_term_already_in_group_uses_existing_variable( def test_derived_bounds_promotes_reference_terms( self, mocker, future_context, standard_name ): + if not standard_name and isinstance(future_context, contextlib.nullcontext): + pytest.skip("Test only applicable when FUTURE context is enabled.") + # CFVariable.attributes is a snapshot taken at construction, so the + # "absent" case is built without standard_name from the start rather + # than deleted from the mock afterwards - a post-construction `del` + # is invisible to a CFVariable built from this mock either way. self.root = netcdf_variable( mocker, "z", "z", np.float64, formula_terms="a: pressure", - standard_name="custom_reference", + standard_name="custom_reference" if standard_name else None, ) - if not standard_name: - if isinstance(future_context, contextlib.nullcontext): - pytest.skip("Test only applicable when FUTURE context is enabled.") - else: - del self.root.standard_name self.pressure = netcdf_variable(mocker, "pressure", "z", np.float64) self.variables = {"z": self.root, "pressure": self.pressure, "temp": self.data} self._patch_encoded_dataset(mocker) @@ -827,6 +915,54 @@ def test_derived_bounds_promotes_reference_terms( # Promotion step is skipped if standard_name is absent. assert "pressure" not in cf_group.promoted + def test_promotes_using_long_name_when_standard_name_empty(self, mocker): + # A present-but-empty standard_name is falsy, so the "or" in + # _reader.py falls through to long_name. Both go in the + # non-derived_bounds branch: under iris.FUTURE.derived_bounds the + # code takes the "continue" at :504 only when standard_name is + # *absent*, never reached here since "" is present, but the + # promotion machinery itself only runs outside that guard. + self.root = netcdf_variable( + mocker, + "z", + "z", + np.float64, + formula_terms="a: pressure", + standard_name="", + long_name="custom_reference", + ) + self.pressure = netcdf_variable(mocker, "pressure", "z", np.float64) + self.variables = {"z": self.root, "pressure": self.pressure, "temp": self.data} + self._patch_encoded_dataset(mocker) + mocker.patch.dict( + "iris.fileformats.cf.reference_terms", + {"custom_reference": "a"}, + clear=False, + ) + + cf_group = CFReader("dummy.nc").cf_group + + assert "pressure" in cf_group.promoted + + def test_raises_when_standard_name_empty_and_long_name_absent(self, mocker): + # The one case ``.get("long_name")`` would survive: standard_name + # present-but-falsy forces the "or" to evaluate long_name, and with + # no long_name at all the subscript must raise KeyError. + self.root = netcdf_variable( + mocker, + "z", + "z", + np.float64, + formula_terms="a: pressure", + standard_name="", + ) + self.pressure = netcdf_variable(mocker, "pressure", "z", np.float64) + self.variables = {"z": self.root, "pressure": self.pressure, "temp": self.data} + self._patch_encoded_dataset(mocker) + + with pytest.raises(KeyError, match="long_name"): + CFReader("dummy.nc") + class Test_translate__global_attributes_missing: @pytest.fixture(autouse=True) @@ -936,3 +1072,53 @@ def test_derived_bounds_boundary_guard_continue_branch(self, mocker, guard_fires assert "orog" not in self.reader.cf_group.promoted else: assert "orog" in self.reader.cf_group.promoted + + +class TestSynthesisedBoundsLink: + """Pin the write-and-read-the-same-place invariant at unit level. + + This class does not exercise `_reader.py` - it builds a + `CFAuxiliaryCoordinateVariable` directly and characterises + `TrackedAttributes` plus `CFVariable.__getattr__`: a write to + ``cf_var.attributes`` is a write to the one place + ``cf_var.`` reads from, including when the written value is + `None`. `_reader.py`'s own synthesised-bounds writes at :327 and :345 + are exercised by + `lib/iris/tests/integration/netcdf/derived_bounds/test_bounds_files.py`, + which fails three ways if either write is removed. + """ + + def test_a_written_bounds_link_is_visible_to_a_reader(self, mocker): + nc_var = mocker.MagicMock() + nc_var.attributes = {"units": "m"} + nc_var.dimensions = ("model_level_number",) + cf_var = CFAuxiliaryCoordinateVariable("a", nc_var) + + assert "bounds" not in cf_var.attributes + cf_var.attributes["bounds"] = "a_bnds" + + assert cf_var.attributes["bounds"] == "a_bnds" + assert "bounds" in cf_var.attributes + assert cf_var.bounds == "a_bnds" + + def test_a_written_bounds_link_overrides_the_file(self, mocker): + nc_var = mocker.MagicMock() + nc_var.attributes = {"bounds": "from_file"} + nc_var.dimensions = ("model_level_number",) + cf_var = CFAuxiliaryCoordinateVariable("a", nc_var) + + cf_var.attributes["bounds"] = "synthesised" + assert cf_var.bounds == "synthesised" + + def test_a_bounds_link_set_to_none_still_reads_as_present(self, mocker): + # _reader.py:358 invalidates a broken link by setting it to None + # rather than deleting it, and the reads downstream check presence + # before value. + nc_var = mocker.MagicMock() + nc_var.attributes = {"bounds": "broken"} + nc_var.dimensions = ("model_level_number",) + cf_var = CFAuxiliaryCoordinateVariable("a", nc_var) + + cf_var.attributes["bounds"] = None + assert "bounds" in cf_var.attributes + assert cf_var.attributes.get("bounds") is None diff --git a/lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py b/lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py new file mode 100644 index 0000000000..a5ec593d2c --- /dev/null +++ b/lib/iris/tests/unit/fileformats/cf/test_CFReader__dataset.py @@ -0,0 +1,223 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Tests for :class:`iris.fileformats.cf.CFReader` reading a real file. + +The rest of the CFReader tests drive it with mocks, which is the right shape +for its classification logic and the wrong shape for the question here: what +type the reader actually hands out, and what a load does with it. These use a +small file on disk instead. +""" + +import numpy as np +import pytest + +import iris +from iris.fileformats.cf import CFReader +from iris.fileformats.cf.dataset import CFDatasetVariable +from iris.fileformats.netcdf import _dataset, _thread_safe_nc + +UNREAD_COMMENT = "an attribute nothing in Iris reads" + + +@pytest.fixture +def sample_path(tmp_path): + """Write a small CF file: a data variable, a coordinate and its bounds.""" + path = tmp_path / "sample.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w") + dataset.title = "a sample file" + dataset.createDimension("time", 3) + dataset.createDimension("bnds", 2) + + time = dataset.createVariable("time", "f8", ("time",)) + time.standard_name = "time" + time.units = "days since 1970-01-01" + time.bounds = "time_bnds" + time.comment = UNREAD_COMMENT + time[:] = np.arange(3, dtype="f8") + + bounds = dataset.createVariable("time_bnds", "f8", ("time", "bnds")) + bounds[:] = np.zeros((3, 2)) + + air = dataset.createVariable("air", "f4", ("time",)) + air.standard_name = "air_temperature" + air.units = "K" + air.coordinates = "time" + air.comment = UNREAD_COMMENT + air[:] = np.arange(3, dtype="f4") + + dataset.close() + return path + + +@pytest.fixture +def char_path(tmp_path): + """Write a small CF file with a character auxiliary coordinate. + + A separate fixture from ``sample_path``: adding a char variable there + would change what ``test_only_unread_attributes_reach_the_cube`` and its + neighbours load. + """ + path = tmp_path / "char.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w") + dataset.createDimension("x", 3) + dataset.createDimension("strlen", 3) + + labels = dataset.createVariable("labels", "S1", ("x", "strlen")) + labels._Encoding = "utf-8" + labels[:] = np.array([list(s) for s in ("aaa", "bbb", "ccc")], dtype="S1") + + temp = dataset.createVariable("temp", "f4", ("x",)) + temp.standard_name = "air_temperature" + temp.units = "K" + temp.coordinates = "labels" + temp[:] = np.arange(3, dtype="f4") + + dataset.close() + return path + + +class TestABorrowedBareDataset: + """A behaviour change from before this branch. + + Before this branch, ``CFReader`` stored a borrowed dataset untouched + (``self._dataset = file_source``), so a bare ``netCDF4.Dataset`` - what + the Xarray bridge hands to ``iris.load``, per ``loader.py``'s docstring - + kept its character data as raw bytes. ``NetCDFDataset.from_existing`` now + wraps anything lacking ``THREAD_SAFE_FLAG`` in an ``EncodedDataset`` + regardless of the caller's own decoding preference, so the same bare + dataset's character data is decoded to Python strings instead. This is + deliberate - the new behaviour is better, and a char auxiliary coordinate + that used to be silently dropped by ``iris.load`` now loads - but it is a + genuine, previously-undocumented change on a public entry point, so this + pins it. + """ + + def test_char_data_is_decoded(self, char_path): + # A genuinely bare, unwrapped dataset - not a DatasetWrapper, not an + # EncodedDataset - is the point of this test. + import netCDF4 + + raw = netCDF4.Dataset(char_path, mode="r") + try: + with CFReader(raw) as reader: + labels = reader.cf_group["labels"] + assert labels.dtype.kind == "U" + assert labels.shape == (3,) + np.testing.assert_array_equal(labels[:], ["aaa", "bbb", "ccc"]) + finally: + raw.close() + + +class TestTheSwap: + def test_the_reader_owns_a_netcdf_dataset(self, sample_path): + with CFReader(str(sample_path)) as reader: + assert isinstance(reader._dataset, _dataset.NetCDFDataset) + + def test_cf_data_is_a_cf_dataset_variable(self, sample_path): + # The whole point of the PR: nothing downstream of here needs to know + # the file is netCDF. + with CFReader(str(sample_path)) as reader: + assert isinstance(reader.cf_group["air"].cf_data, CFDatasetVariable) + + def test_attributes_come_from_the_file(self, sample_path): + with CFReader(str(sample_path)) as reader: + air = reader.cf_group["air"] + assert air.attributes["units"] == "K" + assert air.units == "K" + assert air.dimensions == ("time",) + + def test_global_attributes_come_from_the_dataset(self, sample_path): + with CFReader(str(sample_path)) as reader: + assert reader.cf_group.global_attributes["title"] == "a sample file" + + def test_filename_is_the_variables_location(self, sample_path): + with CFReader(str(sample_path)) as reader: + assert reader.cf_group["air"].filename == str(sample_path) + + def test_the_bounds_variable_was_classified(self, sample_path): + # identify() now reads through .attributes, so a miss here means the + # classification pass lost sight of the file's attributes entirely. + with CFReader(str(sample_path)) as reader: + assert "time_bnds" in reader.cf_group.bounds + + def test_a_borrowed_dataset_is_used_and_not_closed(self, sample_path): + raw = _thread_safe_nc.DatasetWrapper(sample_path, mode="r") + try: + with CFReader(raw) as reader: + assert reader.cf_group["air"].units == "K" + assert raw.isopen() + finally: + raw.close() + + +class TestAttributesAreASnapshot: + """Finding F11. + + ``NetCDFDatasetVariable.attributes`` writes through to the file, which is + right for the saver and wrong for the reader: CFReader synthesises a + "bounds" link during load, on a file it opened read-only. CFVariable + therefore takes a copy. + """ + + def test_a_write_does_not_reach_the_file(self, sample_path): + with CFReader(str(sample_path)) as reader: + air = reader.cf_group["air"] + air.attributes["bounds"] = "invented" + + assert "bounds" not in air.cf_data.attributes + + +class TestAttributeTrackingAcrossReset: + """Review Focus 3. + + CFReader reads attributes while classifying variables, then calls + ``cf_attrs_reset()`` so the rules start from a clean record. Until this + PR, ``__getattr__`` cached the value on the instance, so a read after the + reset found the cache and was never recorded - and an attribute Iris had + in fact consumed was still reported unused, and so was copied onto the + cube as if the file had volunteered it. + """ + + def test_the_reader_resets_what_it_read_while_classifying(self, sample_path): + with CFReader(str(sample_path)) as reader: + time = reader.cf_group["time"] + # CFBoundaryVariable.identify() read "bounds" during __init__. + # Ask through .untracked, so that asking does not itself record. + assert "bounds" in time.attributes.untracked + assert dict(time.cf_attrs_used()) == dict(time.cf_attrs_ignored()) + + def test_a_read_after_the_reset_is_recorded(self, sample_path): + with CFReader(str(sample_path)) as reader: + time = reader.cf_group["time"] + + assert time.units == "days since 1970-01-01" + assert "units" in dict(time.cf_attrs_used()) + assert "units" not in dict(time.cf_attrs_unused()) + + def test_an_unread_attribute_reaches_the_cube(self, sample_path): + # Named for what this pins, not the stronger claim it does not: see + # the comment below on the "only" direction. + cube = iris.load_cube(str(sample_path)) + + # "title" is a global attribute: build_and_add_global_attributes + # copies cf_group.global_attributes onto the cube unconditionally, + # untouched by the per-variable used/unused tracking this test + # otherwise pins - so it is expected here alongside the coordinate + # variable's genuinely-unread "comment". + assert cube.attributes == {"title": "a sample file", "comment": UNREAD_COMMENT} + assert cube.coord("time").attributes == {"comment": UNREAD_COMMENT} + + # Known gap: this does not pin the complementary "only" direction - + # that *no* read attribute reaches the cube. Every attribute this + # fixture's variables read ("units", "standard_name", "coordinates", + # "bounds") is also in saver.py's _CF_ATTRS, so _add_unused_attributes + # (loader.py) filters them out before tracking is ever consulted; + # mutating cf_attrs_unused() to cf_attrs_used() fails this test, but + # mutating it to attributes.untracked.items() - tracking made blind, + # so everything counts as unused - still passes. Pinning "only" would + # need a non-_CF_ATTRS attribute the load rules genuinely read and + # record, e.g. a UGRID or grid-mapping attribute, which means adding + # a mesh or a grid mapping to a fixture whose subject is attribute + # tracking, and would change what the file loads into. diff --git a/lib/iris/tests/unit/fileformats/cf/test_CFVariable.py b/lib/iris/tests/unit/fileformats/cf/test_CFVariable.py index e5489df00e..e5a51e1934 100644 --- a/lib/iris/tests/unit/fileformats/cf/test_CFVariable.py +++ b/lib/iris/tests/unit/fileformats/cf/test_CFVariable.py @@ -4,10 +4,12 @@ # See LICENSE in the root of the repository for full licensing details. """Unit tests for :class:`iris.fileformats.cf.CFVariable`.""" +import numpy as np import pytest from iris.fileformats import cf as cf from iris.fileformats.cf import _variables +from iris.fileformats.cf.dataset import TrackedAttributes class CFVariableSub(cf.CFVariable): @@ -19,18 +21,15 @@ def identify(self, variables, ignore=None, target=None, warn=True): def make_nc_var(mocker): nc_var = mocker.MagicMock() - nc_var.ncattrs.return_value = ["coordinates", "standard_name", "_FillValue"] - nc_var.getncattr.side_effect = { + nc_var.attributes = { "coordinates": "x y", "standard_name": "air_temperature", "_FillValue": -999, - }.__getitem__ - nc_var.coordinates = "x y" - nc_var.standard_name = "air_temperature" + } nc_var.dimensions = ("time", "lat") + nc_var.location = "/tmp/file.nc" nc_var.__len__.return_value = 4 nc_var.__getitem__.return_value = "payload" - nc_var.group.return_value.filepath.return_value = "/tmp/file.nc" return nc_var @@ -41,8 +40,8 @@ def nc_var(mocker): @pytest.fixture -def nc_var_without_group(nc_var): - del nc_var.group +def nc_var_without_location(nc_var): + del nc_var.location return nc_var @@ -54,7 +53,7 @@ def nc_vars(mocker): class TestInit: - def test_records_filename_from_group(self, nc_var): + def test_records_filename_from_the_variables_location(self, nc_var): cf_var = CFVariableSub("foo", nc_var) assert cf_var.filename == "/tmp/file.nc" @@ -64,8 +63,10 @@ def test_records_filename_from_group(self, nc_var): assert cf_var.cf_terms_by_root == {} assert cf_var._to_be_promoted is False - def test_falls_back_to_unknown_filename_without_group(self, nc_var_without_group): - cf_var = CFVariableSub("foo", nc_var_without_group) + def test_falls_back_to_unknown_filename_without_a_location( + self, nc_var_without_location + ): + cf_var = CFVariableSub("foo", nc_var_without_location) assert cf_var.filename == "" @@ -115,6 +116,59 @@ def test_is_subset_check(self, mocker, nc_vars): other = CFVariableSub("other", other_nc_var) assert not other.spans(rhs) + def test_scalar_dimension_with_others_is_not_scalar(self, mocker, nc_var): + # _scalar_ means "this variable is scalar", which a variable with a + # second dimension is not. It is not a wildcard. + nc_var.dimensions = (_variables._NCZARR_SCALAR_DIMENSION, "bnds") + cf_var = CFVariableSub("not_scalar", nc_var) + + other = mocker.MagicMock() + other.dimensions = ("time",) + + assert not cf_var.spans(other) + + def test_no_dimensions_always_spans(self, mocker, nc_var): + nc_var.dimensions = () + cf_var = CFVariableSub("scalar", nc_var) + + other = mocker.MagicMock() + other.dimensions = ("time",) + + assert cf_var.spans(other) + + +class TestIsScalar: + """Direct tests of the single definition of "scalar" that spans() shares. + + _is_scalar() is this task's produced interface: PR 4's Zarr backend + reads it rather than restating the rule, so it is tested directly and + not only through spans(). + """ + + def test_no_dimensions_is_scalar(self, nc_var): + nc_var.dimensions = () + cf_var = CFVariableSub("scalar", nc_var) + + assert cf_var._is_scalar() + + def test_nczarr_scalar_dimension_is_scalar(self, nc_var): + nc_var.dimensions = (_variables._NCZARR_SCALAR_DIMENSION,) + cf_var = CFVariableSub("scalar", nc_var) + + assert cf_var._is_scalar() + + def test_nczarr_scalar_dimension_with_another_is_not_scalar(self, nc_var): + nc_var.dimensions = (_variables._NCZARR_SCALAR_DIMENSION, "bnds") + cf_var = CFVariableSub("not_scalar", nc_var) + + assert not cf_var._is_scalar() + + def test_ordinary_dimensions_are_not_scalar(self, nc_var): + nc_var.dimensions = ("time", "lat") + cf_var = CFVariableSub("not_scalar", nc_var) + + assert not cf_var._is_scalar() + class TestComparisonAndRepresentation: def test_equality_inequality_and_hash_by_name(self, nc_vars): @@ -135,41 +189,50 @@ def test_repr_contains_class_name_name_and_data_repr(self, nc_var): class TestAttributeAccess: - def test_cached(self, nc_var): - # Make sure attribute access to the underlying netCDF4.Variable - # is cached. - name = "foo" - cf_var = CFVariableSub(name, nc_var) - assert nc_var.ncattrs.call_count == 1 - - # Accessing a netCDF attribute should result in no further calls - # to nc_var.ncattrs() and the creation of an attribute on the - # cf_var. - # NB. Can't use hasattr() because that triggers the attribute - # to be created! + def test_reads_are_not_cached_on_the_instance(self, nc_var): + # The setattr cache is gone. It made a re-read after cf_attrs_reset() + # invisible to attribute tracking, which decided which attributes + # reached the cube - see TestReadAfterReset below. + cf_var = CFVariableSub("foo", nc_var) + + assert "coordinates" not in cf_var.__dict__ + assert cf_var.coordinates == "x y" + assert "coordinates" not in cf_var.__dict__ + assert cf_var.coordinates == "x y" assert "coordinates" not in cf_var.__dict__ - _ = cf_var.coordinates - assert nc_var.ncattrs.call_count == 1 - assert "coordinates" in cf_var.__dict__ - # Trying again results in no change. - _ = cf_var.coordinates - assert nc_var.ncattrs.call_count == 1 - assert "coordinates" in cf_var.__dict__ + def test_attributes_are_a_snapshot_taken_at_construction(self, nc_var): + # Finding F11. Copied, not wrapped: the storage object's mapping + # writes through to the file, and CFReader synthesises a "bounds" + # link on a file it opened read-only. + cf_var = CFVariableSub("foo", nc_var) - # Trying another attribute results in just a new attribute. - assert "standard_name" not in cf_var.__dict__ - _ = cf_var.standard_name - assert nc_var.ncattrs.call_count == 1 - assert "standard_name" in cf_var.__dict__ + nc_var.attributes["units"] = "K" + assert "units" not in cf_var.attributes.untracked + + cf_var.attributes["bounds"] = "foo_bnds" + assert "bounds" not in nc_var.attributes - def test_getattr_non_ncattr_value_is_cached_but_not_marked_used(self, nc_var): - nc_var.not_an_ncattr = 42 + def test_getattr_of_a_non_attribute_reaches_the_variable(self, nc_var): + # The one-cycle compatibility route, now delegated to the storage + # object - which is also what warns. See Step 15. + nc_var.deprecated_netcdf_member.return_value = 42 cf_var = CFVariableSub("foo", nc_var) assert cf_var.not_an_ncattr == 42 - assert "not_an_ncattr" in cf_var.__dict__ - assert "not_an_ncattr" not in cf_var.cf_attrs() + nc_var.deprecated_netcdf_member.assert_called_once_with("not_an_ncattr") + # __getattr__ must not cache the value onto the instance - a second + # access has to reach the storage object again, not a stale copy. + assert "not_an_ncattr" not in cf_var.__dict__ + assert "not_an_ncattr" not in dict(cf_var.cf_attrs()) + assert "not_an_ncattr" not in dict(cf_var.cf_attrs_used()) + + def test_getattr_of_nothing_at_all_raises_attribute_error(self, nc_var): + nc_var.deprecated_netcdf_member.side_effect = AttributeError("nonesuch") + cf_var = CFVariableSub("foo", nc_var) + + with pytest.raises(AttributeError, match="nonesuch"): + cf_var.nonesuch def test_getitem_and_len_delegate_to_underlying_variable(self, nc_var): cf_var = CFVariableSub("foo", nc_var) @@ -216,3 +279,246 @@ def test_subclass_stub_returns_none(self, nc_var): cf_var = CFVariableSub("foo", nc_var) assert cf_var.identify({}) is None + + +class TestAttributesMapping: + def test_attributes_contents(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + assert dict(cf_var.attributes) == { + "coordinates": "x y", + "standard_name": "air_temperature", + "_FillValue": -999, + } + + def test_getattr_and_mapping_are_the_same_read(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + _ = cf_var.coordinates + assert cf_var.attributes.read == frozenset(["_FillValue", "coordinates"]) + + cf_var.cf_attrs_reset() + _ = cf_var.attributes["coordinates"] + assert cf_var.attributes.read == frozenset(["_FillValue", "coordinates"]) + + def test_untracked_read_is_not_recorded(self, nc_var): + # What helpers.py's flag-attribute probe needs: a look that does not + # count, so the attribute still reaches the cube. + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.attributes.untracked["coordinates"] == "x y" + assert "coordinates" in dict(cf_var.cf_attrs_unused()) + + +class TestTypedProperties: + def test_dimensions_is_a_tuple(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + assert cf_var.dimensions == ("time", "lat") + + def test_shape_ndim_dtype_size(self, mocker, nc_var): + nc_var.shape = (3, 4) + nc_var.dtype = np.dtype("f4") + nc_var.size = 12 + cf_var = CFVariableSub("foo", nc_var) + + assert cf_var.shape == (3, 4) + assert cf_var.ndim == 2 + assert cf_var.dtype == np.dtype("f4") + assert cf_var.size == 12 + + #: Each typed property, against the value its netCDF4 variable supplies. + TYPED_PROPERTIES = { + "dimensions": ("time", "lat"), + "shape": (3, 4), + "ndim": 2, + "dtype": np.dtype("f4"), + "size": 12, + } + + @pytest.mark.parametrize("name", list(TYPED_PROPERTIES)) + def test_a_file_attribute_colliding_with_a_typed_property(self, nc_var, name): + # A deliberate, accepted difference from pre-PR behaviour. The old + # __getattr__ recorded such a name as read before forwarding to the + # netCDF4 variable, so _add_unused_attributes dropped it and a + # legitimate user attribute silently vanished from the loaded cube - + # even though the value returned was never the file's. A declared + # property now short-circuits __getattr__ entirely, nothing is + # recorded, and the attribute survives onto the cube. + nc_var.attributes = {name: f"file value of {name}"} + nc_var.shape = (3, 4) + nc_var.dtype = np.dtype("f4") + nc_var.size = 12 + cf_var = CFVariableSub("foo", nc_var) + + # The netCDF4 value wins, not the colliding file attribute. + assert getattr(cf_var, name) == self.TYPED_PROPERTIES[name] + + # Reading the property recorded nothing, so the file attribute is + # still unused - and so still reaches the cube - with its own value + # intact. Deliberately asserted through cf_attrs_unused() rather than + # cf_var.attributes[name], which would itself record the read this + # test exists to detect. + assert dict(cf_var.cf_attrs_unused())[name] == f"file value of {name}" + + +#: CFVariable members that a CF attribute of the same name cannot displace. +SHADOWED_NAMES = ["filename", "cf_name", "spans", "attributes", "cf_data"] + + +class TestShadowedAttributeNames: + """Review Focus 1. Spec section 4.3's known limitation, pinned.""" + + @pytest.fixture + def shadowing(self, nc_var): + nc_var.attributes = { + name: f"file value of {name}" for name in SHADOWED_NAMES + } | {"units": "K"} + return CFVariableSub("foo", nc_var) + + def test_the_class_member_wins(self, nc_var, shadowing): + # Each member's own value, not merely "not the file value": that + # weaker form passes against a member returning None or another + # member's value, and is trivially true of attributes and cf_data, + # neither of which can ever equal a string. + assert shadowing.filename == "/tmp/file.nc" + assert shadowing.cf_name == "foo" + assert shadowing.spans.__func__ is CFVariableSub.spans + assert isinstance(shadowing.attributes, TrackedAttributes) + assert shadowing.cf_data is nc_var + + # Fails if SHADOWED_NAMES gains a member this test does not assert. + assert set(SHADOWED_NAMES) == { + "filename", + "cf_name", + "spans", + "attributes", + "cf_data", + } + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_the_file_value_is_still_reachable(self, shadowing, name): + assert shadowing.attributes[name] == f"file value of {name}" + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_reading_it_through_the_mapping_marks_it_used(self, shadowing, name): + assert name in dict(shadowing.cf_attrs_unused()) + _ = shadowing.attributes[name] + assert name in dict(shadowing.cf_attrs_used()) + assert name not in dict(shadowing.cf_attrs_unused()) + + @pytest.mark.parametrize("name", SHADOWED_NAMES) + def test_it_is_listed_among_the_variables_attributes(self, shadowing, name): + assert name in dict(shadowing.cf_attrs()) + + def test_a_shadowing_name_does_not_disturb_its_neighbours(self, shadowing): + assert shadowing.units == "K" + assert dict(shadowing.cf_attrs_used())["units"] == "K" + + +class TestGetattrAndHasattr: + """Review Focus 2. The two call shapes the loading rules actually use.""" + + def test_getattr_with_a_default_finds_the_attribute(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + assert getattr(cf_var, "coordinates", None) == "x y" + + def test_getattr_with_a_default_returns_the_default(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + assert getattr(cf_var, "nonesuch", None) is None + assert getattr(cf_var, "nonesuch", "fallback") == "fallback" + + def test_a_missing_attribute_raises_attribute_error_not_key_error(self, nc_var): + # getattr(..., default) only swallows AttributeError. A KeyError from + # the mapping would escape and abort the load. + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + with pytest.raises(AttributeError): + cf_var.nonesuch + + def test_getattr_of_a_default_does_not_record_a_read(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + _ = getattr(cf_var, "nonesuch", None) + assert cf_var.cf_attrs_used() == (("_FillValue", -999),) + + def test_hasattr_true_marks_the_attribute_used(self, nc_var): + # Parity with the old behaviour: hasattr went through __getattr__, + # which added the name to the used set. + cf_var = CFVariableSub("foo", nc_var) + assert hasattr(cf_var, "coordinates") + assert "coordinates" in dict(cf_var.cf_attrs_used()) + + def test_hasattr_false_marks_nothing(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + cf_var.cf_data = object() + assert not hasattr(cf_var, "nonesuch") + assert cf_var.cf_attrs_used() == (("_FillValue", -999),) + + def test_dunder_probes_do_not_reach_the_file(self, nc_var): + # copy, pickle and numpy all probe for dunders. Answering one from + # file data would make a CFVariable behave as whatever the file says. + nc_var.attributes = {"__array__": "nonsense"} + cf_var = CFVariableSub("foo", nc_var) + + assert not hasattr(cf_var, "__array_interface__") + # The file really does carry "__array__", and __getattr__ still + # refuses it - without that refusal numpy would believe the file. + with pytest.raises(AttributeError, match="__array__"): + cf_var.__array__ + assert cf_var.attributes.untracked["__array__"] == "nonsense" + + @pytest.mark.parametrize("name", ["attributes", "cf_data"]) + def test_the_recursion_guard_holds_before_init_completes(self, name): + # An instance whose __init__ never ran - what copy and pickle build - + # must raise, not recurse until the stack is gone. + cf_var = CFVariableSub.__new__(CFVariableSub) + with pytest.raises(AttributeError, match=name): + getattr(cf_var, name) + + +class TestReadAfterReset: + """Review Focus 3. The cache removal's whole point, at unit scale. + + Task 11, Step 1 pins the same behaviour end to end, on a real file. + """ + + def test_a_re_read_after_reset_counts_again(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + + # As CFReader does: read while parsing structure, then reset. + _ = cf_var.coordinates + assert "coordinates" in dict(cf_var.cf_attrs_used()) + cf_var.cf_attrs_reset() + assert "coordinates" in dict(cf_var.cf_attrs_unused()) + + # As the loading rules then do: read it again. + _ = cf_var.coordinates + assert "coordinates" in dict(cf_var.cf_attrs_used()) + assert "coordinates" not in dict(cf_var.cf_attrs_unused()) + + def test_the_same_holds_through_hasattr(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + _ = cf_var.coordinates + cf_var.cf_attrs_reset() + + assert hasattr(cf_var, "coordinates") + assert "coordinates" in dict(cf_var.cf_attrs_used()) + + def test_reset_restores_the_ignored_names_only(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + _ = cf_var.coordinates + _ = cf_var.standard_name + cf_var.cf_attrs_reset() + + assert cf_var.cf_attrs_used() == (("_FillValue", -999),) + assert cf_var.cf_attrs_unused() == ( + ("coordinates", "x y"), + ("standard_name", "air_temperature"), + ) + + def test_values_are_unchanged_by_any_of_this(self, nc_var): + cf_var = CFVariableSub("foo", nc_var) + for _ in range(3): + assert cf_var.coordinates == "x y" + cf_var.cf_attrs_reset() diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/__init__.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/__init__.py index 36ea8c9953..ae387631e2 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/__init__.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/__init__.py @@ -10,6 +10,10 @@ import pytest from pytest_mock import MockerFixture +from iris.fileformats.cf import CFDataVariable +from iris.fileformats.cf._variables import _CF_ATTRS_IGNORE +from iris.fileformats.cf.dataset import TrackedAttributes + class MockerMixin: mocker: MockerFixture @@ -17,3 +21,99 @@ class MockerMixin: @pytest.fixture(autouse=True) def _mocker_mixin_setup(self, mocker): self.mocker = mocker + + +class CFVariableDouble: + """A minimal stand-in for :class:`iris.fileformats.cf.CFVariable`. + + The loading rules read a CF variable's file attributes through exactly + two things: a real :class:`~iris.fileformats.cf.dataset.TrackedAttributes` + mapping at ``.attributes``, and ``CFVariable.__getattr__`` routing unknown + names to it. This double models both, so a test can read + ``double.some_name`` to build its own expected value and the production + code's ``double.attributes.get("some_name")`` resolves the identical + value from the same underlying mapping - unlike a flat + ``Mock(some_name=...)``, which has nothing behind ``.attributes`` at all. + + Every keyword given becomes a CF attribute: present in ``.attributes`` + and returned via plain attribute access (through ``__getattr__``). A name + not given raises ``AttributeError`` on plain access and, through + ``.attributes.get``, returns the caller's default - exactly as a real + missing file attribute would. Structural members that production + ``CFVariable`` exposes as plain instance attributes (``cf_name``, + ``cf_group``, ``cf_data``, ``filename``, ...) are not modelled here; + set them directly on the instance after construction, as real + ``CFVariable`` does outside ``.attributes``. + """ + + def __init__(self, **attributes): + # Seed with the same ignored names production CFVariable.__init__ + # does: scale_factor, add_offset and friends must start already-read + # here too, or a test that puts one into a double sees it as unread + # (and so, e.g., surviving onto a built object's attributes) when a + # real load never would. + self.attributes = TrackedAttributes(dict(attributes), ignored=_CF_ATTRS_IGNORE) + + def __getattr__(self, name): + # Mirrors production CFVariable.__getattr__'s recursion guard + # (_variables.py:86,275): "attributes" is what this method reads to + # resolve anything else, so it must never be resolved by this method + # itself. Without this, an instance with no "attributes" yet in its + # __dict__ - e.g. `cls.__new__(cls)` during copy/deepcopy/unpickling, + # then a `__setstate__` probe - recurses infinitely instead of + # raising AttributeError. + if name.startswith("__") or name == "attributes": + raise AttributeError(name) + try: + return self.attributes[name] + except KeyError: + raise AttributeError(name) from None + + def __getitem__(self, key): + """Index the double's data, exactly as ``CFVariable.__getitem__`` does. + + Real ``CFVariable.__getitem__`` returns ``self.cf_data[key]``; a test + that needs indexing sets ``.cf_data`` to an indexable data array, the + same way it sets any other structural member. + """ + return self.cf_data[key] + + def cf_attrs(self): + """Return all attribute name/value pairs, exactly as ``CFVariable`` does. + + Used by the "last resort" raw-cube fallback (``build_raw_cube``), so a + double that hits that path still works without further setup. + """ + attributes = self.attributes.untracked + return tuple((name, attributes[name]) for name in sorted(attributes)) + + +class RealArrayCfData: + """Wrap a real (non-lazy) array as a :class:`CFVariableDouble`'s ``cf_data``. + + ``_get_cf_var_data`` reads ``cf_data.is_emulated`` and + ``cf_data.is_variable_length`` before it ever looks at size, so a double + backed directly by a plain array - as most of these tests are, since the + array given is the coordinate's real data rather than something read from + a file - needs this much of the storage interface even though the array + is always far too small to reach ``.chunking``/``.variable``, the two + members only the lazy-loading branch reads. + """ + + is_emulated = False + is_variable_length = False + + def __init__(self, array): + self._array = array + + def __getitem__(self, key): + return self._array[key] + + +# A handful of call sites assert `isinstance(cf_var, CFDataVariable)` as a +# sanity check on their own caller (e.g. `helpers.get_attr_units`, invoked +# with `capture_invalid=True` only when building a Cube's own units). Register +# the double as a virtual subclass so that check passes without inheriting +# CFDataVariable's construction or behaviour - a double is not a real +# CFDataVariable, but the loading rules only ever probe it with `isinstance`. +CFDataVariable.register(CFVariableDouble) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__add_or_capture.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__add_or_capture.py index cadd2efa62..d2386b7cf9 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__add_or_capture.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__add_or_capture.py @@ -9,8 +9,8 @@ from iris.cube import Cube from iris.fileformats._nc_load_rules import helpers -from iris.fileformats.cf import CFVariable from iris.loading import LOAD_PROBLEMS, LoadProblems +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble class Mixin: @@ -30,9 +30,8 @@ def make_args(self, mocker): self.build_func = mocker.MagicMock() self.build_func.return_value = "BUILT" self.add_method = mocker.MagicMock() - self.cf_var = mocker.MagicMock(spec=CFVariable) + self.cf_var = CFVariableDouble(**{self.attr_key: self.attr_value}) self.cf_var.filename = self.filename - setattr(self.cf_var, self.attr_key, self.attr_value) def call( self, diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__normalise_bounds_units.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__normalise_bounds_units.py index 4d85ccacd6..9747defaa7 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__normalise_bounds_units.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test__normalise_bounds_units.py @@ -4,18 +4,20 @@ # See LICENSE in the root of the repository for full licensing details. """Test function :func:`iris.fileformats._nc_load_rules.helpers._normalise_bounds_units`.""" -from typing import Optional +from typing import Any import numpy as np import pytest -from pytest_mock import MockType from iris.fileformats._nc_load_rules.helpers import ( _normalise_bounds_units, _WarnComboIgnoringCfLoad, ) from iris.tests import _shared_utils -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) from iris.warnings import IrisCfLoadWarning CF_NAME = "dummy_bnds" @@ -27,38 +29,40 @@ def _setup(self): self.bounds = self.mocker.sentinel.bounds def _make_cf_bounds_var( - self, - units: Optional[str] = None, - unitless: bool = False, - ) -> MockType: - """Construct a mock CF bounds variable.""" + self, units: str | None = None, unitless: bool = False + ) -> CFVariableDouble: + """Construct a double CF bounds variable. + + Deliberately no ``flag_values``/``flag_masks``/``flag_meanings``: their + absence from ``.attributes`` is what tells ``helpers.get_attr_units`` + this is not a flag variable. + """ if units is None: units = "days since 1970-01-01" - cf_data = self.mocker.Mock(spec=[]) - # we want to mock the absence of flag attributes to helpers.get_attr_units - # see https://docs.python.org/3/library/unittest.mock.html#deleting-attributes - del cf_data.flag_values - del cf_data.flag_masks - del cf_data.flag_meanings - - cf_var = self.mocker.MagicMock( - cf_name=CF_NAME, - cf_data=cf_data, - units=units, - calendar=None, - dtype=float, - ) - - if unitless: - del cf_var.units + attrs: dict[str, Any] = {"calendar": None} + if not unitless: + attrs["units"] = units + cf_var = CFVariableDouble(**attrs) + # cf_name/dtype are structural members of a real CFVariable, not CF + # attributes, so CFVariableDouble doesn't declare them - mypy can't + # see that setting them here is exactly what the docstring prescribes. + cf_var.cf_name = CF_NAME # type: ignore[attr-defined] + cf_var.dtype = float # type: ignore[attr-defined] return cf_var def test_unitless(self) -> None: """Test bounds variable with no units.""" cf_bounds_var = self._make_cf_bounds_var(unitless=True) - result = _normalise_bounds_units(None, cf_bounds_var, self.bounds) + # cf_bounds_var is deliberately a CFVariableDouble, not a real + # CFBoundaryVariable - see the class docstring. Repeated below at + # every other call for the same reason. + result = _normalise_bounds_units( + None, + cf_bounds_var, # type: ignore[arg-type] + self.bounds, + ) assert result == self.bounds def test_invalid_units__pass_through(self) -> None: @@ -67,7 +71,11 @@ def test_invalid_units__pass_through(self) -> None: cf_bounds_var = self._make_cf_bounds_var(units=units) wmsg = f"Ignoring invalid units {units!r} on netCDF variable {CF_NAME!r}" with pytest.warns(_WarnComboIgnoringCfLoad, match=wmsg): - result = _normalise_bounds_units(None, cf_bounds_var, self.bounds) + result = _normalise_bounds_units( + None, + cf_bounds_var, # type: ignore[arg-type] + self.bounds, + ) assert result == self.bounds @pytest.mark.parametrize("units", ["unknown", "no_unit", "1", "kelvin"]) @@ -80,7 +88,11 @@ def test_ignore_bounds(self, units) -> None: f"Expected units compatible with {points_units!r}" ) with pytest.warns(IrisCfLoadWarning, match=wmsg): - result = _normalise_bounds_units(points_units, cf_bounds_var, self.bounds) + result = _normalise_bounds_units( + points_units, + cf_bounds_var, # type: ignore[arg-type] + self.bounds, + ) assert result is None def test_compatible(self) -> None: @@ -88,7 +100,11 @@ def test_compatible(self) -> None: points_units, bounds_units = "days since 1970-01-01", "hours since 1970-01-01" cf_bounds_var = self._make_cf_bounds_var(units=bounds_units) bounds = np.arange(10, dtype=float) * 24 - result = _normalise_bounds_units(points_units, cf_bounds_var, bounds) + result = _normalise_bounds_units( + points_units, + cf_bounds_var, # type: ignore[arg-type] + bounds, + ) expected = bounds / 24 _shared_utils.assert_array_equal(result, expected) @@ -97,6 +113,10 @@ def test_same_units(self) -> None: units = "days since 1970-01-01" cf_bounds_var = self._make_cf_bounds_var(units=units) bounds = np.arange(10, dtype=float) - result = _normalise_bounds_units(units, cf_bounds_var, bounds) + result = _normalise_bounds_units( + units, + cf_bounds_var, # type: ignore[arg-type] + bounds, + ) _shared_utils.assert_array_equal(result, bounds) assert result is bounds diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_albers_equal_area_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_albers_equal_area_coordinate_system.py index d9a33dd948..fe1a077b5a 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_albers_equal_area_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_albers_equal_area_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_albers_equal_area_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildAlbersEqualAreaCoordinateSystem(MockerMixin): @@ -52,7 +55,7 @@ def _test(self, inverse_flattening=False, no_optionals=False): gridvar_props["semi_minor_axis"] = 6356256.909 expected_ellipsoid = iris.coord_systems.GeogCS(6377563.396, 6356256.909) - cf_grid_var = self.mocker.Mock(spec=[], **gridvar_props) + cf_grid_var = CFVariableDouble(**gridvar_props) cs = build_albers_equal_area_coordinate_system(None, cf_grid_var) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_ancil_var.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_ancil_var.py index ce2fd7bf8b..0abdf66906 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_ancil_var.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_ancil_var.py @@ -11,8 +11,11 @@ from iris.cube import Cube from iris.exceptions import CannotAddError from iris.fileformats._nc_load_rules.helpers import build_and_add_ancil_var -from iris.fileformats.cf import CFAncillaryDataVariable from iris.loading import LOAD_PROBLEMS +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + RealArrayCfData, +) @pytest.fixture @@ -28,22 +31,16 @@ def mock_engine(mocker): @pytest.fixture def mock_cf_av_var(mocker, monkeypatch, mock_engine): data = np.arange(6) - output = mocker.Mock( - spec=CFAncillaryDataVariable, - dimensions=("foo",), - scale_factor=1, - add_offset=0, - cf_name="wibble", - cf_data=mocker.MagicMock(chunking=mocker.Mock(return_value=None), spec=[]), - filename=mock_engine.filename, - standard_name=None, - long_name="wibble", - units="m2", - shape=data.shape, - size=np.prod(data.shape), - dtype=data.dtype, - __getitem__=lambda self, key: data[key], - ) + output = CFVariableDouble(standard_name=None, long_name="wibble", units="m2") + output.dimensions = ("foo",) + output.scale_factor = 1 + output.add_offset = 0 + output.cf_name = "wibble" + output.cf_data = RealArrayCfData(data) + output.filename = mock_engine.filename + output.shape = data.shape + output.size = np.prod(data.shape) + output.dtype = data.dtype # Create patch for deferred loading that prevents attempted # file access. This assumes that output is defined in the test case. diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_auxiliary_coordinate.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_auxiliary_coordinate.py index 0d4b16e8da..784e0057b7 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_auxiliary_coordinate.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_auxiliary_coordinate.py @@ -17,9 +17,12 @@ from iris.cube import Cube from iris.exceptions import CannotAddError from iris.fileformats._nc_load_rules.helpers import build_and_add_auxiliary_coordinate -from iris.fileformats.cf import CFVariable from iris.loading import LOAD_PROBLEMS -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, + RealArrayCfData, +) class TestBoundsVertexDim(MockerMixin): @@ -48,20 +51,16 @@ def _setup(self, mocker): cube_parts=dict(coordinates=[]), ) - self.cf_coord_var = mocker.Mock( - spec=CFVariable, - dimensions=dimension_names, - cf_name="wibble", - cf_data=cf_data, - filename=self.engine.filename, - standard_name=None, - long_name="wibble", - units="km", - shape=points.shape, - size=np.prod(points.shape), - dtype=points.dtype, - __getitem__=lambda self, key: points[key], + self.cf_coord_var = CFVariableDouble( + standard_name=None, long_name="wibble", units="km" ) + self.cf_coord_var.dimensions = dimension_names + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = RealArrayCfData(points) + self.cf_coord_var.filename = self.engine.filename + self.cf_coord_var.shape = points.shape + self.cf_coord_var.size = np.prod(points.shape) + self.cf_coord_var.dtype = points.dtype expected_bounds, _ = self._make_array_and_cf_data( mocker, dimension_names=("foo", "bar", "nv") @@ -119,18 +118,14 @@ def _make_cf_bounds_var(self, mocker, dimension_names, rollaxis=False): mocker, dimension_names, rollaxis=rollaxis ) bounds *= 1000 # Convert to metres. - cf_bounds_var = self.mocker.Mock( - spec=CFVariable, - dimensions=dimension_names, - cf_name="wibble_bnds", - cf_data=cf_data, - filename=self.engine.filename, - units="m", - shape=bounds.shape, - size=np.prod(bounds.shape), - dtype=bounds.dtype, - __getitem__=lambda self, key: bounds[key], - ) + cf_bounds_var = CFVariableDouble(units="m") + cf_bounds_var.dimensions = dimension_names + cf_bounds_var.cf_name = "wibble_bnds" + cf_bounds_var.cf_data = RealArrayCfData(bounds) + cf_bounds_var.filename = self.engine.filename + cf_bounds_var.shape = bounds.shape + cf_bounds_var.size = np.prod(bounds.shape) + cf_bounds_var.dtype = bounds.dtype return cf_bounds_var @@ -169,8 +164,20 @@ class TestDtype(MockerMixin): def _setup(self, mocker): # Create coordinate cf variables and pyke engine. points = np.arange(6).reshape(2, 3) - cf_data = mocker.MagicMock(_FillValue=None, shape=points.shape) - cf_data.chunking = mocker.MagicMock(return_value=points.shape) + # `cf_data` must support both `.chunking` (the lazy-loading path, + # forced on by `deferred_load_patch` below, which also needs + # `.variable` for the proxy it builds) and real indexing (via + # `CFVariableDouble.__getitem__`, which reads `self.cf_data[key]` + # exactly as production `CFVariable.__getitem__` does). + cf_data = mocker.MagicMock( + _FillValue=None, + shape=points.shape, + is_emulated=False, + is_variable_length=False, + chunking=points.shape, + variable=mocker.MagicMock(shape=points.shape), + ) + cf_data.__getitem__ = mocker.MagicMock(side_effect=lambda key: points[key]) self.engine = mocker.Mock( cube=mocker.Mock(), @@ -179,20 +186,16 @@ def _setup(self, mocker): cube_parts=dict(coordinates=[]), ) - self.cf_coord_var = mocker.Mock( - spec=CFVariable, - dimensions=("foo", "bar"), - cf_name="wibble", - cf_data=cf_data, - filename=self.engine.filename, - standard_name=None, - long_name="wibble", - units="m", - shape=points.shape, - size=np.prod(points.shape), - dtype=points.dtype, - __getitem__=lambda self, key: points[key], + self.cf_coord_var = CFVariableDouble( + standard_name=None, long_name="wibble", units="m" ) + self.cf_coord_var.dimensions = ("foo", "bar") + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = cf_data + self.cf_coord_var.filename = self.engine.filename + self.cf_coord_var.shape = points.shape + self.cf_coord_var.size = np.prod(points.shape) + self.cf_coord_var.dtype = points.dtype @contextlib.contextmanager def deferred_load_patch(self): @@ -211,8 +214,8 @@ def patched__getitem__(proxy_self, keys): yield def test_scale_factor_add_offset_int(self): - self.cf_coord_var.scale_factor = 3 - self.cf_coord_var.add_offset = 5 + self.cf_coord_var.attributes["scale_factor"] = 3 + self.cf_coord_var.attributes["add_offset"] = 5 build_and_add_auxiliary_coordinate(self.engine, self.cf_coord_var) @@ -220,7 +223,7 @@ def test_scale_factor_add_offset_int(self): assert coord.dtype.kind == "i" def test_scale_factor_float(self): - self.cf_coord_var.scale_factor = 3.0 + self.cf_coord_var.attributes["scale_factor"] = 3.0 with self.deferred_load_patch(): build_and_add_auxiliary_coordinate(self.engine, self.cf_coord_var) @@ -229,7 +232,7 @@ def test_scale_factor_float(self): assert coord.dtype.kind == "f" def test_add_offset_float(self): - self.cf_coord_var.add_offset = 5.0 + self.cf_coord_var.attributes["add_offset"] = 5.0 with self.deferred_load_patch(): build_and_add_auxiliary_coordinate(self.engine, self.cf_coord_var) @@ -251,44 +254,35 @@ def _setup(self, mocker): points = np.arange(6) units = "days since 1970-01-01" - self.cf_coord_var = mocker.Mock( - spec=CFVariable, - dimensions=("foo",), - scale_factor=1, - add_offset=0, - cf_name="wibble", - cf_data=mocker.MagicMock(chunking=mocker.Mock(return_value=None), spec=[]), - filename=self.engine.filename, + self.cf_coord_var = CFVariableDouble( standard_name=None, long_name="wibble", units=units, calendar=None, - shape=points.shape, - size=np.prod(points.shape), - dtype=points.dtype, - __getitem__=lambda self, key: points[key], ) + self.cf_coord_var.dimensions = ("foo",) + self.cf_coord_var.scale_factor = 1 + self.cf_coord_var.add_offset = 0 + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = RealArrayCfData(points) + self.cf_coord_var.filename = self.engine.filename + self.cf_coord_var.shape = points.shape + self.cf_coord_var.size = np.prod(points.shape) + self.cf_coord_var.dtype = points.dtype bounds = np.arange(12).reshape(6, 2) - cf_data = mocker.MagicMock(chunking=mocker.Mock(return_value=None)) - # we want to mock the absence of flag attributes to helpers.get_attr_units - # see https://docs.python.org/3/library/unittest.mock.html#deleting-attributes - del cf_data.flag_values - del cf_data.flag_masks - del cf_data.flag_meanings - self.cf_bounds_var = mocker.Mock( - spec=CFVariable, - dimensions=("x", "nv"), - scale_factor=1, - add_offset=0, - cf_name="wibble_bnds", - cf_data=cf_data, - units=units, - shape=bounds.shape, - size=np.prod(bounds.shape), - dtype=bounds.dtype, - __getitem__=lambda self, key: bounds[key], - ) + # No flag_values/flag_masks/flag_meanings: their absence from + # `.attributes` is what tells helpers.get_attr_units this is not a + # flag variable. + self.cf_bounds_var = CFVariableDouble(units=units) + self.cf_bounds_var.dimensions = ("x", "nv") + self.cf_bounds_var.scale_factor = 1 + self.cf_bounds_var.add_offset = 0 + self.cf_bounds_var.cf_name = "wibble_bnds" + self.cf_bounds_var.cf_data = RealArrayCfData(bounds) + self.cf_bounds_var.shape = bounds.shape + self.cf_bounds_var.size = np.prod(bounds.shape) + self.cf_bounds_var.dtype = bounds.dtype self.bounds = bounds # Create patch for deferred loading that prevents attempted diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_measure.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_measure.py index aea5061f1e..a528d7b58a 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_measure.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_measure.py @@ -11,8 +11,11 @@ from iris.cube import Cube from iris.exceptions import CannotAddError from iris.fileformats._nc_load_rules.helpers import build_and_add_cell_measure -from iris.fileformats.cf import CFMeasureVariable from iris.loading import LOAD_PROBLEMS +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + RealArrayCfData, +) @pytest.fixture @@ -28,23 +31,17 @@ def mock_engine(mocker): @pytest.fixture def mock_cf_cm_var(monkeypatch, mock_engine, mocker): data = np.arange(6) - output = mocker.Mock( - spec=CFMeasureVariable, - dimensions=("foo",), - scale_factor=1, - add_offset=0, - cf_name="wibble", - cf_data=mocker.MagicMock(chunking=mocker.Mock(return_value=None), spec=[]), - filename=mock_engine.filename, - standard_name=None, - long_name="wibble", - units="m2", - shape=data.shape, - size=np.prod(data.shape), - dtype=data.dtype, - __getitem__=lambda self, key: data[key], - cf_measure="area", - ) + output = CFVariableDouble(standard_name=None, long_name="wibble", units="m2") + output.dimensions = ("foo",) + output.scale_factor = 1 + output.add_offset = 0 + output.cf_name = "wibble" + output.cf_data = RealArrayCfData(data) + output.filename = mock_engine.filename + output.shape = data.shape + output.size = np.prod(data.shape) + output.dtype = data.dtype + output.cf_measure = "area" # Create patch for deferred loading that prevents attempted # file access. This assumes that output is defined in the test case. diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_methods.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_methods.py index 31c063a90d..61627c6e17 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_methods.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_cell_methods.py @@ -9,18 +9,16 @@ from iris.coords import CellMethod from iris.cube import Cube from iris.fileformats._nc_load_rules import helpers -from iris.fileformats.cf import CFDataVariable from iris.loading import LOAD_PROBLEMS +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble @pytest.fixture def mock_cf_data_var(mocker): - return mocker.Mock( - spec=CFDataVariable, - cell_methods="time: mean", - cf_name="wibble", - filename="DUMMY", - ) + double = CFVariableDouble(cell_methods="time: mean") + double.cf_name = "wibble" + double.filename = "DUMMY" + return double @pytest.fixture diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py index 7e3155ad68..e6360895a9 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_dimension_coordinate.py @@ -15,7 +15,11 @@ from iris.exceptions import CannotAddError from iris.fileformats._nc_load_rules.helpers import build_and_add_dimension_coordinate from iris.loading import LOAD_PROBLEMS -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, + RealArrayCfData, +) class RulesTestMixin(MockerMixin): @@ -57,24 +61,16 @@ def get_cf_bounds_var(coord_var): def _make_bounds_var(self, bounds, dimensions, units): bounds = np.array(bounds) - cf_data = self.mocker.Mock(spec=[]) - # we want to mock the absence of flag attributes to helpers.get_attr_units - # see https://docs.python.org/3/library/unittest.mock.html#deleting-attributes - del cf_data.flag_values - del cf_data.flag_masks - del cf_data.flag_meanings - result = self.mocker.Mock( - dimensions=dimensions, - cf_name="wibble_bnds", - cf_data=cf_data, - units=units, - calendar=None, - shape=bounds.shape, - size=np.prod(bounds.shape), - dtype=bounds.dtype, - __getitem__=lambda self, key: bounds[key], - ) - delattr(result, "_data_array") + # No flag_values/flag_masks/flag_meanings: their absence from + # `.attributes` is what tells helpers.get_attr_units this is not a + # flag variable. + result = CFVariableDouble(units=units, calendar=None) + result.dimensions = dimensions + result.cf_name = "wibble_bnds" + result.cf_data = RealArrayCfData(bounds) + result.shape = bounds.shape + result.size = np.prod(bounds.shape) + result.dtype = bounds.dtype return result @@ -94,21 +90,19 @@ def _setup(self): self.monkeypatch = pytest.MonkeyPatch() def _set_cf_coord_var(self, points): - self.cf_coord_var = self.mocker.Mock( - dimensions=("foo",), - cf_name="wibble", - cf_data=self.mocker.Mock(spec=[]), + self.cf_coord_var = CFVariableDouble( standard_name=None, long_name="wibble", units="days since 1970-01-01", calendar=None, - shape=points.shape, - size=np.prod(points.shape), - dtype=points.dtype, - __getitem__=lambda self, key: points[key], - cf_attrs=lambda: [("foo", "a"), ("bar", "b")], ) - delattr(self.cf_coord_var, "_data_array") + self.cf_coord_var.dimensions = ("foo",) + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = RealArrayCfData(points) + self.cf_coord_var.shape = points.shape + self.cf_coord_var.size = np.prod(points.shape) + self.cf_coord_var.dtype = points.dtype + self.cf_coord_var.filename = "DUMMY" def check_case_dim_coord_construction(self, climatology=False): # Test a generic dimension coordinate, with or without @@ -346,17 +340,14 @@ class TestBoundsVertexDim(RulesTestMixin): def _setup(self, mocker): # Create test coordinate cf variable. points = np.arange(6) - self.cf_coord_var = mocker.Mock( - dimensions=("foo",), - cf_name="wibble", - standard_name=None, - long_name="wibble", - cf_data=mocker.Mock(spec=[]), - units="km", - shape=points.shape, - dtype=points.dtype, - __getitem__=lambda self, key: points[key], + self.cf_coord_var = CFVariableDouble( + standard_name=None, long_name="wibble", units="km" ) + self.cf_coord_var.dimensions = ("foo",) + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = RealArrayCfData(points) + self.cf_coord_var.shape = points.shape + self.cf_coord_var.dtype = points.dtype def test_slowest_varying_vertex_dim__normalise_bounds(self): # Create the bounds cf variable. @@ -442,17 +433,14 @@ def _setup(self): def _make_vars(self, points, bounds=None, units="degrees"): points = np.array(points) - self.cf_coord_var = self.mocker.MagicMock( - dimensions=("foo",), - cf_name="wibble", - standard_name=None, - long_name="wibble", - cf_data=self.mocker.Mock(spec=[]), - units=units, - shape=points.shape, - dtype=points.dtype, - __getitem__=lambda self, key: points[key], + self.cf_coord_var = CFVariableDouble( + standard_name=None, long_name="wibble", units=units ) + self.cf_coord_var.dimensions = ("foo",) + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = RealArrayCfData(points) + self.cf_coord_var.shape = points.shape + self.cf_coord_var.dtype = points.dtype if bounds: bounds = np.array(bounds).reshape(self.cf_coord_var.shape + (2,)) dimensions = ("x", "nv") @@ -535,17 +523,14 @@ def _make_vars(self, bounds): # the cf var is (), rather than (1,). points = np.array([0.0]) units = "degrees" - self.cf_coord_var = self.mocker.Mock( - dimensions=(), - cf_name="wibble", - standard_name=None, - long_name="wibble", - units=units, - cf_data=self.mocker.Mock(spec=[]), - shape=(), - dtype=points.dtype, - __getitem__=lambda self, key: points[key], + self.cf_coord_var = CFVariableDouble( + standard_name=None, long_name="wibble", units=units ) + self.cf_coord_var.dimensions = () + self.cf_coord_var.cf_name = "wibble" + self.cf_coord_var.cf_data = RealArrayCfData(points) + self.cf_coord_var.shape = () + self.cf_coord_var.dtype = points.dtype bounds = np.array(bounds) dimensions = ("bnds",) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_names.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_names.py index ba6a289a08..826136baa7 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_names.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_names.py @@ -10,6 +10,7 @@ from iris.cube import Cube from iris.fileformats._nc_load_rules.helpers import build_and_add_names from iris.loading import LOAD_PROBLEMS +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble @pytest.fixture @@ -19,15 +20,16 @@ def mock_engine(mocker): "comment": "Mocked test object", } cf_group = mocker.Mock(global_attributes=global_attributes) - cf_var = mocker.MagicMock( - cf_name="wibble", + cf_var = CFVariableDouble( standard_name=None, long_name=None, units="m", - dtype=np.float64, cell_methods=None, - cf_group=cf_group, ) + cf_var.cf_name = "wibble" + cf_var.dtype = np.float64 + cf_var.cf_group = cf_group + cf_var.filename = "foo.nc" engine = mocker.Mock(cube=Cube([23]), cf_var=cf_var, filename="foo.nc") return engine @@ -46,8 +48,8 @@ def check_cube_names(self, inputs, expected): # Expected - The expected cube attributes. exp_standard_name, exp_long_name = expected - self.engine.cf_var.standard_name = standard_name - self.engine.cf_var.long_name = long_name + self.engine.cf_var.attributes["standard_name"] = standard_name + self.engine.cf_var.attributes["long_name"] = long_name # engine = _make_engine(standard_name=standard_name, long_name=long_name) build_and_add_names(self.engine) self.cf_name = self.engine.cf_var.cf_name diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_units.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_units.py index ef2931963f..31d873151b 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_units.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_and_add_units.py @@ -9,20 +9,17 @@ from iris.cube import Cube from iris.fileformats._nc_load_rules import helpers -from iris.fileformats.cf import CFDataVariable from iris.loading import LOAD_PROBLEMS +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble @pytest.fixture def mock_cf_data_var(mocker): - return mocker.Mock( - spec=CFDataVariable, - units="kelvin", - cf_name="wibble", - filename="DUMMY", - dtype=float, - cf_data=mocker.Mock(spec=[]), - ) + double = CFVariableDouble(units="kelvin") + double.cf_name = "wibble" + double.filename = "DUMMY" + double.dtype = float + return double @pytest.fixture @@ -41,7 +38,7 @@ def test_construction(mock_engine): def test_invalid_units(mock_engine, mock_cf_data_var): - mock_cf_data_var.units = "not_built" + mock_cf_data_var.attributes["units"] = "not_built" helpers.build_and_add_units(mock_engine) assert mock_engine.cube.attributes["invalid_units"] == "not_built" load_problem = LOAD_PROBLEMS.problems[-1] diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_geostationary_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_geostationary_coordinate_system.py index d810152196..5e2aa2cc08 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_geostationary_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_geostationary_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_geostationary_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildGeostationaryCoordinateSystem(MockerMixin): @@ -49,7 +52,7 @@ def _test(self, inverse_flattening=False, replace_props=None, remove_props=None) cf_grid_var_kwargs = non_ellipsoid_kwargs.copy() cf_grid_var_kwargs.update(ellipsoid_kwargs) - cf_grid_var = self.mocker.Mock(spec=[], **cf_grid_var_kwargs) + cf_grid_var = CFVariableDouble(**cf_grid_var_kwargs) cs = build_geostationary_coordinate_system(None, cf_grid_var) ellipsoid = iris.coord_systems.GeogCS(**ellipsoid_kwargs) expected = Geostationary(ellipsoid=ellipsoid, **non_ellipsoid_kwargs) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_azimuthal_equal_area_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_azimuthal_equal_area_coordinate_system.py index 0d4af82889..822fae62c9 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_azimuthal_equal_area_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_azimuthal_equal_area_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_lambert_azimuthal_equal_area_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildLambertAzimuthalEqualAreaCoordinateSystem(MockerMixin): @@ -49,7 +52,7 @@ def _test(self, inverse_flattening=False, no_optionals=False): gridvar_props["semi_minor_axis"] = 6356256.909 expected_ellipsoid = iris.coord_systems.GeogCS(6377563.396, 6356256.909) - cf_grid_var = self.mocker.Mock(spec=[], **gridvar_props) + cf_grid_var = CFVariableDouble(**gridvar_props) cs = build_lambert_azimuthal_equal_area_coordinate_system(None, cf_grid_var) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_conformal_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_conformal_coordinate_system.py index 9d91c2a3f7..1a56f242a1 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_conformal_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_lambert_conformal_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_lambert_conformal_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildLambertConformalCoordinateSystem(MockerMixin): @@ -52,7 +55,7 @@ def _test(self, inverse_flattening=False, no_optionals=False): gridvar_props["semi_minor_axis"] = 6356256.909 expected_ellipsoid = iris.coord_systems.GeogCS(6377563.396, 6356256.909) - cf_grid_var = self.mocker.Mock(spec=[], **gridvar_props) + cf_grid_var = CFVariableDouble(**gridvar_props) cs = build_lambert_conformal_coordinate_system(None, cf_grid_var) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_mercator_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_mercator_coordinate_system.py index 3752337ea9..17d35033e4 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_mercator_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_mercator_coordinate_system.py @@ -10,12 +10,12 @@ import iris from iris.coord_systems import Mercator from iris.fileformats._nc_load_rules.helpers import build_mercator_coordinate_system +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble class TestBuildMercatorCoordinateSystem: def test_valid(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, semi_major_axis=6377563.396, semi_minor_axis=6356256.909, @@ -34,8 +34,7 @@ def test_valid(self, mocker): assert cs == expected def test_inverse_flattening(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, semi_major_axis=6377563.396, inverse_flattening=299.3249646, @@ -55,8 +54,7 @@ def test_inverse_flattening(self, mocker): assert cs == expected def test_longitude_missing(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( semi_major_axis=6377563.396, inverse_flattening=299.3249646, standard_parallel=10, @@ -74,8 +72,7 @@ def test_longitude_missing(self, mocker): assert cs == expected def test_standard_parallel_missing(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, semi_major_axis=6377563.396, semi_minor_axis=6356256.909, @@ -92,8 +89,7 @@ def test_standard_parallel_missing(self, mocker): assert cs == expected def test_scale_factor_at_projection_origin(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, semi_major_axis=6377563.396, semi_minor_axis=6356256.909, diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_oblique_mercator_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_oblique_mercator_coordinate_system.py index ed4395fffa..b5447d1a41 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_oblique_mercator_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_oblique_mercator_coordinate_system.py @@ -14,6 +14,7 @@ from iris.fileformats._nc_load_rules.helpers import ( build_oblique_mercator_coordinate_system, ) +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble class ParamTuple(NamedTuple): @@ -148,7 +149,7 @@ def make_variant_inputs(self, request) -> None: self.coord_system_args_expected = list(coord_system_kwargs_expected.values()) def test_attributes(self, mocker): - cf_var_mock = mocker.Mock(spec=[], **self.nc_attributes) + cf_var_mock = CFVariableDouble(**self.nc_attributes) coord_system_mock = mocker.Mock(spec=self.expected_class) setattr(coord_systems, self.expected_class.__name__, coord_system_mock) @@ -163,6 +164,6 @@ def test_deprecation(mocker): longitude_of_projection_origin=0.0, scale_factor_at_projection_origin=1.0, ) - cf_var_mock = mocker.Mock(spec=[], **nc_attributes) + cf_var_mock = CFVariableDouble(**nc_attributes) with pytest.warns(IrisDeprecation, match="azimuth_of_central_line = 90"): _ = build_oblique_mercator_coordinate_system(None, cf_var_mock) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_polar_stereographic_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_polar_stereographic_coordinate_system.py index 3e9396cca4..05c6fff8da 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_polar_stereographic_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_polar_stereographic_coordinate_system.py @@ -12,12 +12,12 @@ from iris.fileformats._nc_load_rules.helpers import ( build_polar_stereographic_coordinate_system, ) +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble class TestBuildPolarStereographicCoordinateSystem: def test_valid_north(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, scale_factor_at_projection_origin=1, @@ -40,8 +40,7 @@ def test_valid_north(self, mocker): assert cs == expected def test_valid_south(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=-90, scale_factor_at_projection_origin=1, @@ -64,8 +63,7 @@ def test_valid_south(self, mocker): assert cs == expected def test_valid_with_standard_parallel(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, standard_parallel=30, @@ -86,8 +84,7 @@ def test_valid_with_standard_parallel(self, mocker): assert cs == expected def test_valid_with_false_easting_northing(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, scale_factor_at_projection_origin=1, @@ -114,8 +111,7 @@ def test_valid_with_false_easting_northing(self, mocker): assert cs == expected def test_valid_nonzero_veritcal_lon(self, mocker): - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=30, latitude_of_projection_origin=90, scale_factor_at_projection_origin=1, diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_raw_cube.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_raw_cube.py index 88b3909592..ae1428fd84 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_raw_cube.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_raw_cube.py @@ -15,8 +15,13 @@ def _make_array_and_cf_data(mocker, dim_lens: dict[str, int]): shape = list(dim_lens.values()) - cf_data = mocker.MagicMock(_FillValue=None, spec=[]) - cf_data.chunking = mocker.MagicMock(return_value=shape) + # This data is small enough that _get_cf_var_data never reaches + # .chunking or .variable - only the two gates it always checks first. + cf_data = mocker.MagicMock( + spec=["is_emulated", "is_variable_length"], + is_emulated=False, + is_variable_length=False, + ) data = np.arange(np.prod(shape), dtype=float) data = data.reshape(shape) return data, cf_data diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_stereographic_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_stereographic_coordinate_system.py index 481d4441f8..e12e1fc55f 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_stereographic_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_stereographic_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_stereographic_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildStereographicCoordinateSystem(MockerMixin): @@ -44,7 +47,7 @@ def _test(self, inverse_flattening=False, no_offsets=False): test_easting = 0 test_northing = 0 - cf_grid_var = self.mocker.Mock(spec=[], **gridvar_props) + cf_grid_var = CFVariableDouble(**gridvar_props) cs = build_stereographic_coordinate_system(None, cf_grid_var) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_transverse_mercator_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_transverse_mercator_coordinate_system.py index f63402dcc2..ec7af3bbab 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_transverse_mercator_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_transverse_mercator_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_transverse_mercator_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildTransverseMercatorCoordinateSystem(MockerMixin): @@ -46,7 +49,7 @@ def _test(self, inverse_flattening=False, no_options=False): test_northing = 0 test_scale_factor = 1.0 - cf_grid_var = self.mocker.Mock(spec=[], **gridvar_props) + cf_grid_var = CFVariableDouble(**gridvar_props) cs = build_transverse_mercator_coordinate_system(None, cf_grid_var) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_verticalp_coordinate_system.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_verticalp_coordinate_system.py index 932e1d085d..e8716ce3aa 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_verticalp_coordinate_system.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_build_verticalp_coordinate_system.py @@ -12,7 +12,10 @@ from iris.fileformats._nc_load_rules.helpers import ( build_vertical_perspective_coordinate_system, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestBuildVerticalPerspectiveCoordinateSystem(MockerMixin): @@ -23,7 +26,6 @@ def _test(self, inverse_flattening=False, no_offsets=False): test_easting = 100.0 test_northing = 200.0 cf_grid_var_kwargs = { - "spec": [], "latitude_of_projection_origin": 1.0, "longitude_of_projection_origin": 2.0, "perspective_point_height": 2000000.0, @@ -45,7 +47,7 @@ def _test(self, inverse_flattening=False, no_offsets=False): test_easting = 0 test_northing = 0 - cf_grid_var = self.mocker.Mock(**cf_grid_var_kwargs) + cf_grid_var = CFVariableDouble(**cf_grid_var_kwargs) ellipsoid = iris.coord_systems.GeogCS(**ellipsoid_kwargs) cs = build_vertical_perspective_coordinate_system(None, cf_grid_var) diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_attr_units.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_attr_units.py index 50698e72f8..6a44787ec9 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_attr_units.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_attr_units.py @@ -7,14 +7,17 @@ """ +import cf_units import numpy as np import pytest from iris.fileformats._nc_load_rules.helpers import get_attr_units -from iris.fileformats.cf import CFDataVariable from iris.loading import LOAD_PROBLEMS from iris.tests import _shared_utils -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) from iris.warnings import IrisCfLoadWarning @@ -25,18 +28,16 @@ def _make_cf_var(self, global_attributes=None): cf_group = self.mocker.Mock(global_attributes=global_attributes) - cf_var = self.mocker.MagicMock( - spec=CFDataVariable, - cf_name="sound_frequency", - cf_data=self.mocker.Mock(spec=[]), - filename="DUMMY", + cf_var = CFVariableDouble( standard_name=None, long_name=None, units="\u266b", - dtype=np.float64, cell_methods=None, - cf_group=cf_group, ) + cf_var.cf_name = "sound_frequency" + cf_var.filename = "DUMMY" + cf_var.dtype = np.float64 + cf_var.cf_group = cf_group return cf_var def test_unicode_character(self): @@ -67,3 +68,22 @@ def test_capture(self): load_problem = LOAD_PROBLEMS.problems[-1] assert load_problem.loaded == {"units": "\u266b"} + + def test_flag_values_untracked(self): + """Presence of flag_values must force NO_UNIT_STRING without recording a read. + + This is the load-bearing assertion for the untracked probe in + `get_attr_units` (`name in cf_var.attributes.untracked`): if that + probe is ever simplified to `name in cf_var.attributes` (a tracked + read), `flag_values`/`flag_masks`/`flag_meanings` would stop being + copied onto loaded cubes by `_add_unused_attributes`, silently. The + second assertion below is the one that catches that regression - the + first alone would still pass. + """ + cf_var = CFVariableDouble(units="1", flag_values="1, 2, 3") + cf_var.cf_name = "flag_var" + + attr_units = get_attr_units(cf_var, {}) + + assert attr_units == cf_units._NO_UNIT_STRING + assert "flag_values" not in cf_var.attributes.read diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_names.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_names.py index d54aec0692..885495405d 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_names.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_get_names.py @@ -10,7 +10,10 @@ import numpy as np from iris.fileformats._nc_load_rules.helpers import get_names -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class TestGetNames(MockerMixin): @@ -28,15 +31,15 @@ class TestGetNames(MockerMixin): """ def _make_cf_var(self, standard_name, long_name, cf_name): - cf_var = self.mocker.Mock( - cf_name=cf_name, + cf_var = CFVariableDouble( standard_name=standard_name, long_name=long_name, units="degrees", - dtype=np.float64, cell_methods=None, - cf_group=self.mocker.Mock(global_attributes={}), ) + cf_var.cf_name = cf_name + cf_var.dtype = np.float64 + cf_var.cf_group = self.mocker.Mock(global_attributes={}) return cf_var def check_names(self, inputs, expected): diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_mercator_parameters.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_mercator_parameters.py index 8ac08c330c..cdd60dc2e6 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_mercator_parameters.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_mercator_parameters.py @@ -11,7 +11,10 @@ import warnings from iris.fileformats._nc_load_rules.helpers import has_supported_mercator_parameters -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class _EngineMixin(MockerMixin): @@ -24,8 +27,7 @@ def engine(self, cf_grid_var, cf_name): class TestHasSupportedMercatorParameters(_EngineMixin): def test_valid_base(self, mocker): cf_name = "mercator" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, false_easting=0, false_northing=0, @@ -41,8 +43,7 @@ def test_valid_base(self, mocker): def test_valid_false_easting_northing(self, mocker): cf_name = "mercator" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, false_easting=15, false_northing=10, @@ -58,8 +59,7 @@ def test_valid_false_easting_northing(self, mocker): def test_valid_standard_parallel(self, mocker): cf_name = "mercator" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=-90, false_easting=0, false_northing=0, @@ -75,8 +75,7 @@ def test_valid_standard_parallel(self, mocker): def test_valid_scale_factor(self, mocker): cf_name = "mercator" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=0, false_easting=0, false_northing=0, @@ -94,8 +93,7 @@ def test_invalid_scale_factor_and_standard_parallel(self, mocker): # Scale factor and standard parallel cannot both be specified for # Mercator projections cf_name = "mercator" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( longitude_of_projection_origin=0, false_easting=0, false_northing=0, diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_polar_stereographic_parameters.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_polar_stereographic_parameters.py index 2bfc801af2..837688bbac 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_polar_stereographic_parameters.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_has_supported_polar_stereographic_parameters.py @@ -13,7 +13,10 @@ from iris.fileformats._nc_load_rules.helpers import ( has_supported_polar_stereographic_parameters, ) -from iris.tests.unit.fileformats.nc_load_rules.helpers import MockerMixin +from iris.tests.unit.fileformats.nc_load_rules.helpers import ( + CFVariableDouble, + MockerMixin, +) class _EngineMixin(MockerMixin): @@ -26,8 +29,7 @@ def engine(self, cf_grid_var, cf_name): class TestHasSupportedPolarStereographicParameters(_EngineMixin): def test_valid_base_north(self, mocker): cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, false_easting=0, @@ -44,8 +46,7 @@ def test_valid_base_north(self, mocker): def test_valid_base_south(self, mocker): cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=-90, false_easting=0, @@ -62,8 +63,7 @@ def test_valid_base_south(self, mocker): def test_valid_straight_vertical_longitude(self, mocker): cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=30, latitude_of_projection_origin=90, false_easting=0, @@ -80,8 +80,7 @@ def test_valid_straight_vertical_longitude(self, mocker): def test_valid_false_easting_northing(self, mocker): cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, false_easting=15, @@ -98,8 +97,7 @@ def test_valid_false_easting_northing(self, mocker): def test_valid_standard_parallel(self, mocker): cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, false_easting=0, @@ -116,8 +114,7 @@ def test_valid_standard_parallel(self, mocker): def test_valid_scale_factor(self, mocker): cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, false_easting=0, @@ -136,8 +133,7 @@ def test_invalid_scale_factor_and_standard_parallel(self, mocker): # Scale factor and standard parallel cannot both be specified for # Polar Stereographic projections cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, false_easting=0, @@ -165,8 +161,7 @@ def test_absent_scale_factor_and_standard_parallel(self, mocker): # Scale factor and standard parallel cannot both be specified for # Polar Stereographic projections cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=90, false_easting=0, @@ -192,8 +187,7 @@ def test_invalid_latitude_of_projection_origin(self, mocker): # Scale factor and standard parallel cannot both be specified for # Polar Stereographic projections cf_name = "polar_stereographic" - cf_grid_var = mocker.Mock( - spec=[], + cf_grid_var = CFVariableDouble( straight_vertical_longitude_from_pole=0, latitude_of_projection_origin=45, false_easting=0, diff --git a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_is_lat_lon.py b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_is_lat_lon.py index b4a4b0d2d8..86fb3c00e2 100644 --- a/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_is_lat_lon.py +++ b/lib/iris/tests/unit/fileformats/nc_load_rules/helpers/test_is_lat_lon.py @@ -5,10 +5,10 @@ """Test function :func:`iris.fileformats._nc_load_rules.helpers._is_lat_lon`.""" from iris.fileformats._nc_load_rules import helpers -from iris.fileformats.cf import CFCoordinateVariable +from iris.tests.unit.fileformats.nc_load_rules.helpers import CFVariableDouble -def test_non_string_units(mocker): - cf_var = mocker.MagicMock(spec=CFCoordinateVariable, units=1.0) +def test_non_string_units(): + cf_var = CFVariableDouble(units=1.0) is_lat_lon = helpers._is_lat_lon(cf_var, [], "latitude", "grid_latitude", "y", []) assert is_lat_lon is False diff --git a/lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py b/lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py new file mode 100644 index 0000000000..938797a364 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/dataset/__init__.py @@ -0,0 +1,5 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :mod:`iris.fileformats.netcdf._dataset`.""" diff --git a/lib/iris/tests/unit/fileformats/netcdf/dataset/conftest.py b/lib/iris/tests/unit/fileformats/netcdf/dataset/conftest.py new file mode 100644 index 0000000000..cb4beec456 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/dataset/conftest.py @@ -0,0 +1,77 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""A real netCDF4 file to test the netCDF CFDataset implementation against. + +Real, rather than mocked, because the members being implemented are precisely +the ones whose netCDF4 behaviour is easy to misremember - what ``chunking()`` +returns for a contiguous variable, what ``dimensions`` is for char data, what +``ncattrs()`` includes once ``fill_value=`` has been passed. +""" + +import numpy as np +import pytest + +from iris.fileformats.netcdf import _thread_safe_nc + +#: How the sample file's dimensions are created. None means unlimited. +SAMPLE_DIMENSION_SIZES = {"time": 3, "lat": 4, "nchars": 8, "record": None} + +#: How they should then read back. An unlimited dimension reports the number +#: of records actually written, which here is none. +SAMPLE_DIMENSIONS = {"time": 3, "lat": 4, "nchars": 8, "record": 0} + +#: The sample file's global attributes. +SAMPLE_GLOBALS = {"Conventions": "CF-1.7", "title": "sample"} + +#: The sample file's air_temperature payload. +SAMPLE_AIR = np.arange(12, dtype="f4").reshape(3, 4) + +#: The sample file's label payload, as the strings it encodes. +SAMPLE_LABELS = ["alpha", "beta", "gamma"] + + +@pytest.fixture(scope="session") +def sample_path(tmp_path_factory): + """Write, once per session, a small netCDF4 file and return its path.""" + path = tmp_path_factory.mktemp("cf_dataset") / "sample.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + try: + for name, size in SAMPLE_DIMENSION_SIZES.items(): + dataset.createDimension(name, size) + for name, value in SAMPLE_GLOBALS.items(): + dataset.setncattr(name, value) + + air = dataset.createVariable( + "air_temperature", + "f4", + ("time", "lat"), + fill_value=-999.0, + zlib=True, + chunksizes=(1, 4), + ) + air.setncattr("units", "K") + air.setncattr("standard_name", "air_temperature") + air.setncattr("coordinates", "height") + air[:] = SAMPLE_AIR + + # Scalar, and deliberately unchunked: chunking() answers differently. + height = dataset.createVariable("height", "f4", ()) + height.setncattr("units", "m") + height[:] = 1.5 + + # Char data: the one case where EncodedVariable changes shape, dtype + # and dimensions out from under the wrapper. + label = dataset.createVariable("label", "S1", ("time", "nchars")) + label[:] = np.array([list(f"{text:<8}") for text in SAMPLE_LABELS], dtype="S1") + finally: + dataset.close() + + return path + + +@pytest.fixture +def sample_location(sample_path): + """The sample file's path, as the string a CFDataset reports.""" + return str(sample_path) diff --git a/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py new file mode 100644 index 0000000000..7d1652eda1 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset.py @@ -0,0 +1,273 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.netcdf._dataset.NetCDFDataset`.""" + +import warnings + +import numpy as np +import pytest + +from iris.fileformats.cf.dataset import CFDataset +from iris.fileformats.netcdf import _bytecoding_datasets, _dataset, _thread_safe_nc +from iris.warnings import IrisLoadWarning + +from .conftest import SAMPLE_DIMENSIONS, SAMPLE_GLOBALS + + +@pytest.fixture +def reader(sample_path): + with _dataset.NetCDFDataset(sample_path) as dataset: + yield dataset + + +class TestOpening: + def test_is_a_cf_dataset(self, reader): + assert isinstance(reader, CFDataset) + + def test_location_is_the_string_path(self, reader, sample_location): + assert reader.location == sample_location + + def test_default_mode_is_read(self, reader): + assert reader.mode == "r" + + def test_starts_open(self, reader): + assert reader.closed is False + + def test_close_is_idempotent(self, sample_path): + dataset = _dataset.NetCDFDataset(sample_path) + dataset.close() + assert dataset.closed is True + dataset.close() + assert dataset.closed is True + + def test_context_manager_closes(self, sample_path): + with _dataset.NetCDFDataset(sample_path) as dataset: + assert dataset.closed is False + assert dataset.closed is True + + def test_missing_file_raises(self, tmp_path): + with pytest.raises((FileNotFoundError, OSError)): + _dataset.NetCDFDataset(tmp_path / "nope.nc") + + def test_decoding_setting_chooses_the_wrapper(self, sample_path): + with _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ.context(True): + with _dataset.NetCDFDataset(sample_path) as dataset: + assert isinstance(dataset.dataset, _bytecoding_datasets.EncodedDataset) + with _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ.context(False): + with _dataset.NetCDFDataset(sample_path) as dataset: + assert not isinstance( + dataset.dataset, _bytecoding_datasets.EncodedDataset + ) + assert isinstance(dataset.dataset, _thread_safe_nc.DatasetWrapper) + + +class TestContents: + def test_variables(self, reader, sample_location): + assert sorted(reader.variables) == ["air_temperature", "height", "label"] + air = reader.variables["air_temperature"] + assert isinstance(air, _dataset.NetCDFDatasetVariable) + assert air.name == "air_temperature" + assert air.location == sample_location + + def test_variables_are_built_once(self, reader): + assert ( + reader.variables["air_temperature"] is reader.variables["air_temperature"] + ) + + def test_variables_share_one_write_lock(self, reader): + # _dask_locks.get_worker_lock() returns a FRESH threading.Lock under + # the threaded scheduler, so a per-variable call would hand out locks + # that exclude nothing. Observed through the write handles, because + # the lock is only made when one is asked for - see + # test_materialising_variables_makes_no_write_lock. + air = reader.variables["air_temperature"] + height = reader.variables["height"] + assert air.write_handle().lock is height.write_handle().lock + assert air.write_handle().lock is reader.write_lock + + def test_materialising_variables_makes_no_write_lock(self, reader): + # Making a lock is a write-path act: _dask_locks.get_worker_lock() + # raises DaskSchedulerTypeError for a scheduler the *saver* does not + # support, naming the saver, and a pure load has no quarrel with any + # scheduler. Reading variables and attributes must therefore leave + # the lock unmade. See NetCDFDataset.write_lock. + assert sorted(reader.variables) == ["air_temperature", "height", "label"] + assert dict(reader.attributes) == SAMPLE_GLOBALS + assert reader.variables["air_temperature"].attributes is not None + assert reader._write_lock is None + + def test_dimensions_are_names_and_lengths(self, reader): + assert dict(reader.dimensions) == SAMPLE_DIMENSIONS + + def test_unlimited_dimension_reports_its_current_length(self, reader): + # Nothing was written along "record", so it is zero-length - which is + # what netCDF4 reports, and what saver.py's membership tests need. + assert reader.dimensions["record"] == 0 + assert "record" in reader.dimensions + + def test_global_attributes(self, reader): + assert dict(reader.attributes) == SAMPLE_GLOBALS + + def test_global_attributes_are_built_once(self, reader): + assert reader.attributes is reader.attributes + + +class _EmulatorWithoutIsopen: + """The least a borrowed emulator can be, and no ``isopen()``. + + ``createVariable`` and ``close`` are what + :class:`~iris.fileformats.netcdf._thread_safe_nc.DatasetWrapper` checks + for before agreeing to wrap an object; ``filepath`` is what + ``from_existing`` reads. Nothing here is exercised - the members exist so + that the wrap succeeds. + """ + + def createVariable(self, *args, **kwargs): + raise NotImplementedError + + def close(self): + raise NotImplementedError + + def filepath(self): + return "" + + +class TestBorrowing: + def test_from_existing_wraps_an_open_dataset(self, sample_path, sample_location): + raw = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + assert dataset.dataset is raw + assert dataset.location == sample_location + assert sorted(dataset.variables) == [ + "air_temperature", + "height", + "label", + ] + finally: + raw.close() + + def test_from_existing_does_not_close_what_it_borrowed(self, sample_path): + raw = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + dataset.close() + assert dataset.closed is True + # Still usable: the borrower released nothing. + assert raw.isopen() + finally: + raw.close() + + def test_from_existing_wraps_a_bare_netcdf4_dataset(self, sample_path): + # What the Xarray bridge hands iris.save / CFReader: an object with + # the netCDF4 API but no thread-safe wrapper around it. + import netCDF4 + + raw = netCDF4.Dataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + assert isinstance(dataset.dataset, _bytecoding_datasets.EncodedDataset) + assert dataset.dataset._contained_instance is raw + finally: + raw.close() + + def test_a_borrowed_emulator_without_isopen_reads_as_open(self): + # NetCDFDataset.closed asks the backing object whether it is open, + # and an emulator need not implement isopen() - ncdata's does, so + # nothing else in the suite reaches this branch. Taken to be open: + # the borrower has no better answer, and close() still overrides it. + dataset = _dataset.NetCDFDataset.from_existing(_EmulatorWithoutIsopen()) + assert dataset.closed is False + dataset.close() + assert dataset.closed is True + + def test_from_existing_mode_is_stipulated_not_observed(self, sample_path): + # netCDF4 exposes no public attribute recording a dataset's open + # mode, so from_existing cannot read the mode a borrowed dataset was + # actually opened in - it *always* reports "r+", regardless of the + # mode below. Pins that stipulated value; it is not a description of + # `raw`'s real mode. + raw = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + try: + dataset = _dataset.NetCDFDataset.from_existing(raw) + assert dataset.mode == "r+" + finally: + raw.close() + + +class TestAutoChartostring: + """Iris decodes byte data itself, so netCDF4 must not do it first. + + CFReader turned this off on every dataset it opened (_reader.py:182). + The dataset now does it, so no caller has to remember. + """ + + def test_turned_off_on_an_opened_dataset(self, sample_path, mocker): + spy = mocker.spy(_thread_safe_nc.DatasetWrapper, "set_auto_chartostring") + with _bytecoding_datasets.DECODE_TO_STRINGS_ON_READ.context(False): + with _dataset.NetCDFDataset(sample_path): + pass + spy.assert_called_once_with(mocker.ANY, False) + + def test_turned_off_on_a_borrowed_dataset(self, sample_path, mocker): + # Spied on the class, not the instance: DatasetWrapper's __setattr__ + # forwards instance-level patching to the contained netCDF4 object + # instead of shadowing the method, so mocker.spy(raw, ...) cannot + # observe the call. + raw = _thread_safe_nc.DatasetWrapper(sample_path, mode="r") + spy = mocker.spy(_thread_safe_nc.DatasetWrapper, "set_auto_chartostring") + try: + _dataset.NetCDFDataset.from_existing(raw) + spy.assert_called_once_with(raw, False) + finally: + raw.close() + + def test_an_encoded_dataset_blocks_it_rather_than_forwarding(self, sample_path): + # EncodedDataset does its own decoding, so the call is inert there - + # which is why making it unconditionally is safe. + with _dataset.NetCDFDataset(sample_path) as dataset: + assert isinstance(dataset.dataset, _bytecoding_datasets.EncodedDataset) + with pytest.raises(TypeError, match="not supported"): + dataset.dataset.set_auto_chartostring(True) + + +class TestLegacyFormatWarning: + def test_silent_by_default(self, tmp_path, sample_path): + legacy = _write_netcdf3(tmp_path) + with warnings.catch_warnings(): + warnings.simplefilter("error") + with _dataset.NetCDFDataset(legacy): + pass + + def test_warns_when_asked(self, tmp_path): + legacy = _write_netcdf3(tmp_path) + with pytest.warns(IrisLoadWarning, match="nccopy"): + with _dataset.NetCDFDataset(legacy, warn_legacy_format=True): + pass + + def test_does_not_warn_for_netcdf4(self, sample_path): + with warnings.catch_warnings(): + warnings.simplefilter("error") + with _dataset.NetCDFDataset(sample_path, warn_legacy_format=True): + pass + + def test_warns_for_a_borrowed_dataset_when_asked(self, tmp_path): + legacy = _write_netcdf3(tmp_path) + raw = _thread_safe_nc.DatasetWrapper(legacy, mode="r") + try: + with pytest.warns(IrisLoadWarning, match="nccopy"): + _dataset.NetCDFDataset.from_existing(raw, warn_legacy_format=True) + finally: + raw.close() + + +def _write_netcdf3(tmp_path): + path = tmp_path / "legacy.nc" + dataset = _thread_safe_nc.DatasetWrapper(path, mode="w", format="NETCDF3_CLASSIC") + dataset.createDimension("time", 2) + variable = dataset.createVariable("time", "f4", ("time",)) + variable[:] = np.arange(2, dtype="f4") + dataset.close() + return path diff --git a/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py new file mode 100644 index 0000000000..aaf4fa7139 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDatasetVariable.py @@ -0,0 +1,259 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for :class:`iris.fileformats.netcdf._dataset.NetCDFDatasetVariable`.""" + +from collections.abc import MutableMapping +import warnings + +import numpy as np +import pytest + +from iris._deprecation import IrisDeprecation +from iris.fileformats.cf.dataset import CFDatasetVariable +from iris.fileformats.netcdf import _bytecoding_datasets, _dataset, _thread_safe_nc + +from .conftest import SAMPLE_AIR, SAMPLE_LABELS + + +@pytest.fixture +def raw(sample_path): + dataset = _thread_safe_nc.DatasetWrapper(sample_path, mode="r") + yield dataset + dataset.close() + + +@pytest.fixture +def encoded(sample_path): + dataset = _bytecoding_datasets.EncodedDataset(sample_path, mode="r") + yield dataset + dataset.close() + + +def wrap(dataset, name, location, write_lock_factory=None): + return _dataset.NetCDFDatasetVariable( + dataset.variables[name], location, write_lock_factory=write_lock_factory + ) + + +@pytest.fixture +def air(raw, sample_location): + return wrap(raw, "air_temperature", sample_location) + + +@pytest.fixture +def height(raw, sample_location): + return wrap(raw, "height", sample_location) + + +class TestStorageProperties: + def test_is_a_cf_dataset_variable(self, air): + assert isinstance(air, CFDatasetVariable) + + def test_name(self, air): + assert air.name == "air_temperature" + + def test_location(self, air, sample_location): + assert air.location == sample_location + + def test_dimensions_are_a_tuple(self, air): + # netCDF4 answers with a tuple already, but the contract says tuple + # and CFVariable.spans does set arithmetic on it, so pin it. + assert air.dimensions == ("time", "lat") + assert isinstance(air.dimensions, tuple) + + def test_shape_dtype_size_ndim(self, air): + assert air.shape == (3, 4) + assert air.dtype == np.dtype("f4") + assert air.size == 12 + assert air.ndim == 2 + + def test_len(self, air): + assert len(air) == 3 + + def test_scalar_has_no_len(self, height): + assert height.shape == () + with pytest.raises(TypeError, match="unsized"): + len(height) + + def test_fill_value(self, air): + assert air.fill_value == -999.0 + + def test_no_fill_value_is_none(self, height): + assert height.fill_value is None + + def test_chunking_is_a_tuple(self, air): + # netCDF4 answers with a list; the contract says a shape. + assert air.chunking == (1, 4) + + def test_contiguous_chunking_is_none(self, height): + # netCDF4 answers "contiguous" here, and loader.py already treats + # that and None identically, so both collapse to None. + assert height.chunking is None + + +class TestData: + def test_getitem(self, air): + np.testing.assert_array_equal(air[:], SAMPLE_AIR) + + def test_getitem_indexed(self, air): + np.testing.assert_array_equal(air[0], SAMPLE_AIR[0]) + + +class TestAttributes: + def test_is_a_plain_mutable_mapping(self, air): + # Tracking belongs to CFVariable, not here: a CFDatasetVariable can + # have two CFVariables promoted over it - see finding F1. + assert isinstance(air.attributes, MutableMapping) + assert not hasattr(air.attributes, "read") + + def test_contents(self, air): + assert dict(air.attributes) == { + "_FillValue": -999.0, + "units": "K", + "standard_name": "air_temperature", + "coordinates": "height", + } + + def test_is_built_once(self, air): + assert air.attributes is air.attributes + + def test_values_are_read_once(self, mocker, sample_location): + # Pins the contract _NetCDFAttributes's own docstring states: values + # are read once, at construction, because the interface promises a + # mapping whose keys/items cost nothing. test_is_built_once above + # only pins that the mapping object is cached, not that its values + # are materialised rather than fetched again on every read - call + # counts are the only way to see that. + variable = mocker.Mock() + variable.ncattrs.return_value = ["units"] + variable.getncattr.return_value = "K" + wrapped = _dataset.NetCDFDatasetVariable(variable, sample_location) + _ = wrapped.attributes["units"] + _ = wrapped.attributes["units"] + assert variable.ncattrs.call_count == 1 + assert variable.getncattr.call_count == 1 + + def test_missing_key_raises_key_error(self, air): + with pytest.raises(KeyError, match="nonesuch"): + air.attributes["nonesuch"] + + def test_unreadable_attribute_becomes_empty_string(self, mocker, sample_location): + # ncattrs() can list a name that getncattr then refuses. _getncattr in + # cf/_reader.py tolerated exactly this with a "" default, and that + # tolerance has to survive the move. + # + # A real VariableWrapper cannot be mocked this way: its __setattr__ + # forwards every set to the contained netCDF4 object, so patching + # one of its methods would write a same-named attribute into the + # file instead. A bare stand-in with the two methods + # _NetCDFAttributes actually calls is enough. + variable = mocker.Mock() + variable.ncattrs.return_value = ["units", "broken"] + variable.getncattr.side_effect = lambda name: ( + "m" if name == "units" else _raise(AttributeError(name)) + ) + wrapped = _dataset.NetCDFDatasetVariable(variable, sample_location) + assert wrapped.attributes["broken"] == "" + + +def _raise(exception): + raise exception + + +class _EmulatedVariable: + """A stand-in for the Xarray bridge's own variable object. + + Never a real, file-backed + :class:`~iris.fileformats.netcdf._thread_safe_nc.VariableWrapper`: that + class's ``__setattr__`` forwards every set to the actual netCDF4 object, + so it cannot carry an ad-hoc ``_data_array`` of its own. + """ + + def ncattrs(self): + return [] + + +class TestNetCDFOnlyMembers: + def test_variable_exposes_the_backing_wrapper(self, raw, sample_location): + # raw.variables constructs a fresh VariableWrapper on every access, + # so the check has to compare against the one instance actually + # passed in, not a second, independently fetched wrapper. + variable = raw.variables["air_temperature"] + wrapped = _dataset.NetCDFDatasetVariable(variable, sample_location) + assert wrapped.variable is variable + + def test_not_variable_length(self, air): + assert air.is_variable_length is False + + def test_not_emulated(self, air): + assert air.is_emulated is False + + def test_emulated_data_array_raises_when_not_emulated(self, air): + with pytest.raises(AttributeError, match="_data_array"): + air.emulated_data_array + + def test_setting_emulated_data_array_raises_when_not_emulated(self, air): + # The setter refuses for the same reason the getter does, and the + # consequence of not refusing is worse: VariableWrapper.__setattr__ + # forwards to the contained object, so the set would write a file + # attribute named "_data_array". + with pytest.raises(AttributeError, match="_data_array"): + air.emulated_data_array = np.ones(3) + assert "_data_array" not in air.variable.ncattrs() + + def test_emulated_round_trip(self, sample_location): + # The Xarray bridge hook, issue #4994: an emulating variable carries + # its own array instead of file storage. + variable = _EmulatedVariable() + wrapped = _dataset.NetCDFDatasetVariable(variable, sample_location) + variable._data_array = np.zeros(3) + assert wrapped.is_emulated is True + np.testing.assert_array_equal(wrapped.emulated_data_array, np.zeros(3)) + wrapped.emulated_data_array = np.ones(3) + np.testing.assert_array_equal(variable._data_array, np.ones(3)) + + +class TestDeprecatedNetcdfMember: + def test_a_reach_through_warns(self, air): + with pytest.warns(IrisDeprecation, match="ncattrs"): + names = air.deprecated_netcdf_member("ncattrs")() + assert sorted(names) == [ + "_FillValue", + "coordinates", + "standard_name", + "units", + ] + + def test_a_missing_name_raises_and_does_not_warn(self, air): + # hasattr() probes land here, and a probe that comes back False is + # not a use of anything - warning about it would be noise. The + # underlying netCDF4 variable raises its own AttributeError, which + # does not name the attribute - only the type matters here. + with warnings.catch_warnings(): + warnings.simplefilter("error") + with pytest.raises(AttributeError): + air.deprecated_netcdf_member("no_such_netcdf_member") + + def test_the_message_names_the_replacement(self, air): + with pytest.warns(IrisDeprecation, match=r"cf_data\.variable"): + air.deprecated_netcdf_member("ncattrs") + + +class TestCharacterData: + def test_raw_wrapper_sees_the_char_dimension(self, raw, sample_location): + label = wrap(raw, "label", sample_location) + assert label.dimensions == ("time", "nchars") + assert label.shape == (3, 8) + assert label.dtype == np.dtype("S1") + + def test_encoded_wrapper_sees_strings(self, encoded, sample_location): + # EncodedVariable drops the trailing char dimension and reports a + # string dtype. The CFDatasetVariable must pass that through, not + # reach around it to the contained netCDF4 variable. + label = wrap(encoded, "label", sample_location) + assert label.dimensions == ("time",) + assert label.shape == (3,) + assert label.dtype == np.dtype("U8") + assert [text.strip() for text in label[:]] == SAMPLE_LABELS diff --git a/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py new file mode 100644 index 0000000000..972330b663 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__contract.py @@ -0,0 +1,40 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Run the shared CFDataset contract against the netCDF implementation.""" + +import pytest + +from iris.fileformats.netcdf import _dataset +from iris.tests.unit.fileformats.cf.dataset.contract import ( + CFDatasetContract, + populate, +) + + +class TestNetCDFDatasetContract(CFDatasetContract): + """The netCDF side of the contract. PR 4 adds the Zarr side.""" + + @pytest.fixture + def readable(self, tmp_path): + path = tmp_path / "contract.nc" + with _dataset.NetCDFDataset(path, mode="w", netcdf_format="NETCDF4") as dataset: + populate(dataset) + with _dataset.NetCDFDataset(path) as dataset: + yield dataset + + @pytest.fixture + def writable(self, tmp_path): + dataset = _dataset.NetCDFDataset( + tmp_path / "written.nc", mode="w", netcdf_format="NETCDF4" + ) + yield dataset + dataset.close() + + @pytest.fixture + def reopen(self, tmp_path): + def _reopen(): + return _dataset.NetCDFDataset(tmp_path / "written.nc") + + return _reopen diff --git a/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py new file mode 100644 index 0000000000..fbc282a279 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/dataset/test_NetCDFDataset__write.py @@ -0,0 +1,259 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Unit tests for the write surface of the netCDF CFDataset implementation.""" + +import numpy as np +import pytest + +from iris.fileformats.netcdf import _bytecoding_datasets, _dataset, _thread_safe_nc + + +@pytest.fixture +def path(tmp_path): + return tmp_path / "written.nc" + + +@pytest.fixture +def writer(path): + with _dataset.NetCDFDataset(path, mode="w", netcdf_format="NETCDF4") as dataset: + yield dataset + + +@pytest.fixture +def grid(writer): + """A 2x3 grid, ready for variables to be created against.""" + writer.create_dimension("y", 3) + writer.create_dimension("x", 2) + return writer + + +class TestCreateDimension: + def test_appears_with_its_length(self, writer): + writer.create_dimension("x", 5) + assert writer.dimensions["x"] == 5 + + def test_none_means_unlimited(self, writer): + # saver.py:829 passes None for a dimension the user asked to make + # unlimited. It reads back as zero-length until records are written. + writer.create_dimension("t", None) + assert writer.dimensions["t"] == 0 + assert writer.dataset.dimensions["t"].isunlimited() + + +class TestCreateVariable: + def test_returns_a_dataset_variable(self, grid): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert isinstance(variable, _dataset.NetCDFDatasetVariable) + assert variable.name == "air" + assert variable.dimensions == ("y", "x") + assert variable.shape == (3, 2) + assert variable.dtype == np.dtype("f4") + + def test_appears_in_the_dataset(self, grid): + created = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert grid.variables["air"].name == created.name + + def test_updates_an_already_materialised_variables_cache(self, grid): + # Pins that create_variable keeps an already-materialised .variables + # mapping current, rather than leaving it stale. saver.py's + # variable-naming collision loops (":1566", ":1698", ":1852", + # ":1908", ":2294") read `.variables` repeatedly while creating + # variables, so a stale cache here would let a name collision go + # undetected. + assert "air" not in grid.variables # materialise the cache first + grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert "air" in grid.variables + + def test_dimensions_default_to_scalar(self, writer): + # saver.py:2080 creates a grid-mapping variable with no dimensions at + # all, and passes only a name and a dtype - see finding F4. + variable = writer.create_variable("grid", np.int32) + assert variable.dimensions == () + assert variable.shape == () + + def test_dimensions_may_be_a_list(self, grid): + # saver.py:1938 and :1983 pass a list, not a tuple. + variable = grid.create_variable("air", np.dtype("f4"), ["y", "x"]) + assert variable.dimensions == ("y", "x") + + def test_fill_value_becomes_an_attribute(self, grid): + variable = grid.create_variable( + "air", np.dtype("f4"), ("y", "x"), fill_value=-1.0 + ) + assert variable.fill_value == -1.0 + assert variable.attributes["_FillValue"] == -1.0 + + def test_encoding_is_passed_through(self, grid): + variable = grid.create_variable( + "air", np.dtype("f4"), ("y", "x"), zlib=True, chunksizes=(1, 2) + ) + assert variable.chunking == (1, 2) + + def test_shares_the_datasets_write_lock(self, grid): + # Observed at write time, not at construction time: the variable is + # handed a factory, and the lock is only made when a write handle + # actually asks for it - see NetCDFDataset.write_lock for why a read + # must not provoke one. + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert grid._write_lock is None + assert variable.write_handle().lock is grid.write_lock + + +class TestWritingData: + def test_setitem_round_trip(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + payload = np.arange(6, dtype="f4").reshape(3, 2) + variable[:] = payload + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + np.testing.assert_array_equal(reader.variables["air"][:], payload) + + def test_write_handle_writes_after_the_dataset_is_closed(self, grid, path): + # This is what a Dask worker gets as a da.store target, long after the + # Saver's own handle on the file has gone. + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + handle = variable.write_handle() + grid.close() + + payload = np.arange(6, dtype="f4").reshape(3, 2) + handle[:] = payload + + with _dataset.NetCDFDataset(path) as reader: + np.testing.assert_array_equal(reader.variables["air"][:], payload) + + def test_write_handle_of_an_encoded_variable_encodes(self, grid): + # Renamed from test_write_handle_matches_the_variable_encoding: the + # handle does not match the variable's encoding, it always encodes. + # This is the encoded-variable half of that rule - the dataset was + # opened for writing, so its variables are encoded - and the test + # below is the unencoded half. An unencoded proxy in either case would + # silently skip string encoding. + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + assert isinstance(variable.variable, _bytecoding_datasets.EncodedVariable) + assert isinstance( + variable.write_handle(), _bytecoding_datasets.EncodedNetCDFWriteProxy + ) + + def test_write_handle_of_an_unencoded_variable_still_encodes(self, path): + # This test used to assert the opposite - that an unencoded variable + # yields a plain NetCDFWriteProxy - on the false premise that the + # saver never meets one. It does: NetCDFDataset.from_existing only + # wraps a dataset lacking THREAD_SAFE_FLAG, so a borrowed + # DatasetWrapper keeps plain variables exactly like the one built + # here, and saving into one reaches this path. Writes have no + # selectable string encoding, so the handle must encode regardless of + # how the variable was obtained. + raw = _thread_safe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + try: + raw.createDimension("x", 2) + raw.createVariable("air", "f4", ("x",)) + variable = _dataset.NetCDFDatasetVariable( + raw.variables["air"], str(path), write_lock_factory=None + ) + assert not isinstance( + variable.variable, _bytecoding_datasets.EncodedVariable + ) + assert isinstance( + variable.write_handle(), _bytecoding_datasets.EncodedNetCDFWriteProxy + ) + finally: + raw.close() + + +class TestSyncAndFinalise: + def test_sync_flushes(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable[:] = np.zeros((3, 2), dtype="f4") + grid.sync() + # Readable while the writer is still open, because sync() flushed. + with _dataset.NetCDFDataset(path) as reader: + assert reader.variables["air"].shape == (3, 2) + + def test_finalise_is_a_no_op(self, grid): + assert grid.finalise() is None + assert grid.closed is False + + +class TestAttributeWrites: + def test_variable_attribute_write_through(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["units"] = "K" + assert variable.attributes["units"] == "K" + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.variables["air"].attributes["units"] == "K" + + def test_global_attribute_write_through(self, writer, path): + writer.attributes["Conventions"] = "CF-1.7" + assert writer.attributes["Conventions"] == "CF-1.7" + writer.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.attributes["Conventions"] == "CF-1.7" + + def test_attribute_deletion(self, grid, path): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["units"] = "K" + del variable.attributes["units"] + assert "units" not in variable.attributes + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + assert "units" not in reader.variables["air"].attributes + + def test_ascii_value_is_written_as_bytes(self, grid): + # _bytes_if_ascii, moved here from saver.py. netCDF4 writes a bytes + # value as NC_CHAR; the coercion is what makes every string attribute + # Iris writes take that type, whatever the file format. + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["units"] = "K" + assert variable.variable.getncattr("units") == "K" + assert _dataset._bytes_if_ascii("K") == b"K" + + def test_non_string_values_pass_through(self, grid): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["valid_min"] = np.float32(-3.5) + assert variable.attributes["valid_min"] == np.float32(-3.5) + assert _dataset._bytes_if_ascii(3) == 3 + + +class TestNonAsciiAttributes: + """Review Focus 5. The coercion's except branch, which is the common one + for real-world metadata: degree signs, accented names, superscripts. + """ + + @pytest.mark.parametrize( + "value", + [ + "degC \N{DEGREE SIGN}", + "na\N{LATIN SMALL LETTER I WITH DIAERESIS}ve", + "m s\N{SUPERSCRIPT MINUS}\N{SUPERSCRIPT ONE}", + ], + ) + def test_round_trips_unchanged(self, grid, path, value): + variable = grid.create_variable("air", np.dtype("f4"), ("y", "x")) + variable.attributes["long_name"] = value + assert variable.attributes["long_name"] == value + grid.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.variables["air"].attributes["long_name"] == value + + def test_is_not_coerced_to_bytes(self): + value = "degC \N{DEGREE SIGN}" + assert _dataset._bytes_if_ascii(value) is value + + def test_global_non_ascii_round_trips(self, writer, path): + value = ( + "Produced at M\N{LATIN SMALL LETTER E WITH ACUTE}" + "t\N{LATIN SMALL LETTER E WITH ACUTE}o" + ) + writer.attributes["institution"] = value + writer.close() + + with _dataset.NetCDFDataset(path) as reader: + assert reader.attributes["institution"] == value diff --git a/lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py b/lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py index 94a8e3cfd5..d959771082 100644 --- a/lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py +++ b/lib/iris/tests/unit/fileformats/netcdf/loader/test__get_cf_var_data.py @@ -11,7 +11,7 @@ from iris._lazy_data import _optimum_chunksize import iris.fileformats.cf -from iris.fileformats.netcdf._thread_safe_nc import VLType +import iris.fileformats.netcdf._dataset from iris.fileformats.netcdf.loader import CHUNK_CONTROL, _get_cf_var_data from iris.tests import _shared_utils from iris.tests.unit.fileformats import MockerMixin @@ -27,24 +27,18 @@ def _setup(self): def _make(self, chunksizes=None, shape=None, dtype="i4", **extra_properties): if shape is None: shape = self.shape - dimensions_dict = { - "dim_" + str(x): self.mocker.Mock(size=x) for x in range(len(shape)) - } - dimension_names = list(dimensions_dict.keys()) cf_data = self.mocker.MagicMock( - _FillValue=None, - _Encoding="", - __getitem__="", - dimensions=dimension_names, - # A parent .group() with dimensions is needed to validate the var dims - group=self.mocker.Mock( - return_value=self.mocker.Mock(dimensions=dimensions_dict) - ), - dtype=dtype, + spec=iris.fileformats.netcdf._dataset.NetCDFDatasetVariable, + fill_value=None, + dimensions=tuple("dim_" + str(x) for x in range(len(shape))), shape=shape, - **extra_properties, + chunking=chunksizes, + is_variable_length=False, + is_emulated=False, + # NetCDFDataProxy reads .shape straight off this: the one + # deliberate escape hatch, and the one place a real shape matters. + variable=self.mocker.MagicMock(shape=shape), ) - cf_data.chunking = self.mocker.MagicMock(return_value=chunksizes) if dtype is not str: # for testing VLen str arrays (dtype=`class `) dtype = np.dtype(dtype) cf_var = self.mocker.MagicMock( @@ -55,6 +49,8 @@ def _make(self, chunksizes=None, shape=None, dtype="i4", **extra_properties): cf_name="DUMMY_VAR", shape=shape, size=np.prod(shape), + attributes={}, + dimensions=cf_data.dimensions, **extra_properties, ) cf_var.__getitem__.return_value = self.mocker.sentinel.real_data_accessed @@ -87,21 +83,15 @@ def test_cf_data_chunk_control(self): def test_cf_data_no_chunks(self): # No chunks means chunks are calculated from the array's shape by - # `iris._lazy_data._optimum_chunksize()`. + # `iris._lazy_data._optimum_chunksize()`. This also covers what used + # to be the separate "contiguous" case: NetCDFDatasetVariable.chunking + # already collapses that netCDF spelling of "unchunked" to None. chunks = None cf_var = self._make(chunks) lazy_data = _get_cf_var_data(cf_var) lazy_data_chunks = [c[0] for c in lazy_data.chunks] _shared_utils.assert_array_equal(lazy_data_chunks, self.expected_chunks) - def test_cf_data_contiguous(self): - # Chunks 'contiguous' is equivalent to no chunks. - chunks = "contiguous" - cf_var = self._make(chunks) - lazy_data = _get_cf_var_data(cf_var) - lazy_data_chunks = [c[0] for c in lazy_data.chunks] - _shared_utils.assert_array_equal(lazy_data_chunks, self.expected_chunks) - def test_type__1kf8_is_lazy(self): cf_var = self._make(shape=(1000,), dtype="f8") var_data = _get_cf_var_data(cf_var) @@ -117,47 +107,41 @@ def test_arraytype__100f8_is_real(self, mocker): var_data = _get_cf_var_data(cf_var) assert var_data is mocker.sentinel.real_data_accessed - def test_vltype__1000str_is_lazy(self, mocker): - # Variable length string type - mock_vltype = mocker.Mock(spec=VLType, dtype=str, name="varlen string type") - cf_var = self._make(shape=(1000,), dtype=str, datatype=mock_vltype) + def test_vltype__1000str_is_lazy(self): + cf_var = self._make(shape=(1000,), dtype=str) + cf_var.cf_data.is_variable_length = True var_data = _get_cf_var_data(cf_var) assert isinstance(var_data, da.Array) def test_vltype__1000str_is_real_with_hint(self, mocker): - # Variable length string type with a hint on the array variable length size - mock_vltype = mocker.Mock(spec=VLType, dtype=str, name="varlen string type") - cf_var = self._make(shape=(100,), dtype=str, datatype=mock_vltype) + cf_var = self._make(shape=(100,), dtype=str) + cf_var.cf_data.is_variable_length = True with CHUNK_CONTROL.set("DUMMY_VAR", _vl_hint=1): var_data = _get_cf_var_data(cf_var) assert var_data is mocker.sentinel.real_data_accessed def test_vltype__100str_is_real(self, mocker): - # Variable length string type - mock_vltype = mocker.Mock(spec=VLType, dtype=str, name="varlen string type") - cf_var = self._make(shape=(100,), dtype=str, datatype=mock_vltype) + cf_var = self._make(shape=(100,), dtype=str) + cf_var.cf_data.is_variable_length = True var_data = _get_cf_var_data(cf_var) assert var_data is mocker.sentinel.real_data_accessed - def test_vltype__100str_is_lazy_with_hint(self, mocker): - # Variable length string type with a hint on the array variable length size - mock_vltype = mocker.Mock(spec=VLType, dtype=str, name="varlen string type") - cf_var = self._make(shape=(100,), dtype=str, datatype=mock_vltype) + def test_vltype__100str_is_lazy_with_hint(self): + cf_var = self._make(shape=(100,), dtype=str) + cf_var.cf_data.is_variable_length = True with CHUNK_CONTROL.set("DUMMY_VAR", _vl_hint=50): var_data = _get_cf_var_data(cf_var) assert isinstance(var_data, da.Array) - def test_vltype__100f8_is_lazy(self, mocker): - # Variable length float64 type - mock_vltype = mocker.Mock(spec=VLType, dtype="f8", name="varlen float64 type") - cf_var = self._make(shape=(1000,), dtype="f8", datatype=mock_vltype) + def test_vltype__100f8_is_lazy(self): + cf_var = self._make(shape=(1000,), dtype="f8") + cf_var.cf_data.is_variable_length = True var_data = _get_cf_var_data(cf_var) assert isinstance(var_data, da.Array) def test_vltype__100f8_is_real_with_hint(self, mocker): - # Variable length float64 type with a hint on the array variable length size - mock_vltype = mocker.Mock(spec=VLType, dtype="f8", name="varlen float64 type") - cf_var = self._make(shape=(100,), dtype="f8", datatype=mock_vltype) + cf_var = self._make(shape=(100,), dtype="f8") + cf_var.cf_data.is_variable_length = True with CHUNK_CONTROL.set("DUMMY_VAR", _vl_hint=2): var_data = _get_cf_var_data(cf_var) assert var_data is mocker.sentinel.real_data_accessed @@ -168,17 +152,30 @@ def test_cf_data_emulation(self, mocker, is_realdata, is_string): # Check that a variable emulation object passes its real data directly. # ... or for string data, that it converts it, either lazy or real. shape = (20, 30, 30) - spec = np.ndarray if is_realdata else da.Array + array_spec = np.ndarray if is_realdata else da.Array dtype = np.dtype("S1") if is_string else np.dtype("f4") - emulated_data = mocker.MagicMock(spec=spec, dtype=dtype, shape=shape) - # Make a cf_var with a special extra '_data_array' property. - cf_var = self._make( - chunksizes=None, - dtype=dtype, - _data_array=emulated_data, - shape=shape, - ) + emulated_data = mocker.MagicMock(spec=array_spec, dtype=dtype, shape=shape) + cf_var = self._make(chunksizes=None, dtype=dtype, shape=shape) + cf_var.cf_data.is_emulated = True + cf_var.cf_data.emulated_data_array = emulated_data + if is_string: + # The encoding is described by the file variable beneath the + # emulation, which is 'char' where the data read out is strings. + unencoded = mocker.Mock( + dtype=np.dtype("S1"), + dimensions=("dim_0", "dim_1", "string5"), + # None is what an absent _Encoding attribute yields: the + # default encoding, and no "unsupported encoding" warning. + _Encoding=None, + ) + # .name is a Mock constructor keyword, so it has to be set after. + unencoded.name = "DUMMY_VAR" + unencoded.group.return_value = mocker.Mock( + dimensions={"string5": mocker.Mock(size=5)} + ) + cf_var.cf_data.unencoded_variable = unencoded + # detect that we added a translation to the data ifnbd = "iris.fileformats.netcdf._bytecoding_datasets." mock_decode = mocker.patch(ifnbd + "decode_bytesarray_to_stringarray") diff --git a/lib/iris/tests/unit/fileformats/netcdf/loader/test__load_cube.py b/lib/iris/tests/unit/fileformats/netcdf/loader/test__load_cube.py index cc6e03a2cc..839446ac6f 100644 --- a/lib/iris/tests/unit/fileformats/netcdf/loader/test__load_cube.py +++ b/lib/iris/tests/unit/fileformats/netcdf/loader/test__load_cube.py @@ -51,7 +51,7 @@ def _make(self, names, attrs): cf_group[name] = self.mocker.Mock(cf_attrs_unused=cf_attrs_unused) cf = self.mocker.Mock(cf_group=cf_group) - cf_data = self.mocker.Mock(_FillValue=None) + cf_data = self.mocker.Mock() cf_data.chunking = self.mocker.MagicMock(return_value=shape) cf_var = self.mocker.MagicMock( spec=iris.fileformats.cf.CFVariable, @@ -61,6 +61,7 @@ def _make(self, names, attrs): cf_group=coords, shape=shape, size=np.prod(shape), + attributes={}, ) return cf, cf_var @@ -141,7 +142,7 @@ def _setup(self, mocker): def _make(self, attrs): shape = (1,) cf_attrs_unused = self.mocker.Mock(return_value=attrs) - cf_data = self.mocker.Mock(_FillValue=None) + cf_data = self.mocker.Mock() cf_data.chunking = self.mocker.MagicMock(return_value=shape) cf_var = self.mocker.MagicMock( spec=iris.fileformats.cf.CFVariable, @@ -153,6 +154,7 @@ def _make(self, attrs): cf_attrs_unused=cf_attrs_unused, shape=shape, size=np.prod(shape), + attributes={}, ) return cf_var diff --git a/lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py b/lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py index 3f386238ea..2dcfa3971a 100644 --- a/lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py +++ b/lib/iris/tests/unit/fileformats/netcdf/loader/test__translate_constraints_to_var_callback.py @@ -15,20 +15,25 @@ from iris.tests import _shared_utils +def _data_variable(mocker, name, **attributes): + """A CFDataVariable whose CF attributes are exactly the ones named.""" + return CFDataVariable(name, mocker.MagicMock(attributes=attributes)) + + class Test: @pytest.fixture(autouse=True) def _setup(self, mocker): self.data_variables = [ - CFDataVariable("var1", mocker.MagicMock(standard_name="x_wind")), - CFDataVariable("var2", mocker.MagicMock(standard_name="y_wind")), - CFDataVariable("var1", mocker.MagicMock(long_name="x component of wind")), - CFDataVariable( + _data_variable(mocker, "var1", standard_name="x_wind"), + _data_variable(mocker, "var2", standard_name="y_wind"), + _data_variable(mocker, "var1", long_name="x component of wind"), + _data_variable( + mocker, "var1", - mocker.MagicMock( - standard_name="x_wind", long_name="x component of wind" - ), + standard_name="x_wind", + long_name="x component of wind", ), - CFDataVariable("var1", mocker.MagicMock()), + _data_variable(mocker, "var1"), ] def test_multiple_constraints(self): @@ -38,7 +43,7 @@ def test_multiple_constraints(self): ] callback = _translate_constraints_to_var_callback(constrs) result = [callback(var) for var in self.data_variables] - _shared_utils.assert_array_equal(result, [True, True, False, True, False]) + _shared_utils.assert_array_equal(result, [True, True, True, True, True]) def test_multiple_constraints_invalid(self): constrs = [ @@ -57,12 +62,12 @@ def test_multiple_constraints__multiname(self, mocker): callback = _translate_constraints_to_var_callback(constrs) # Add 2 extra vars: one passes both name checks, and the other does not vars = self.data_variables + [ - CFDataVariable("var1", mocker.MagicMock(standard_name="x_wind")), - CFDataVariable("var1", mocker.MagicMock(standard_name="air_pressure")), + _data_variable(mocker, "var1", standard_name="x_wind"), + _data_variable(mocker, "var1", standard_name="air_pressure"), ] result = [callback(var) for var in vars] _shared_utils.assert_array_equal( - result, [True, True, False, True, False, True, False] + result, [True, True, True, True, True, True, False] ) def test_non_name_constraint(self): @@ -83,13 +88,19 @@ def test_name_constraint_standard_name(self): constr = iris.NameConstraint(standard_name="x_wind") callback = _translate_constraints_to_var_callback(constr) result = [callback(var) for var in self.data_variables] - _shared_utils.assert_array_equal(result, [True, False, False, True, False]) + _shared_utils.assert_array_equal(result, [True, False, True, True, True]) - def test_name_constraint_long_name(self): + def test_name_constraint_long_name(self, mocker): constr = iris.NameConstraint(long_name="x component of wind") callback = _translate_constraints_to_var_callback(constr) - result = [callback(var) for var in self.data_variables] - _shared_utils.assert_array_equal(result, [False, False, True, True, False]) + # The shared vars either match long_name or omit it, and omission is + # permissive - so without a var that CONTRADICTS, this test would pass + # even if the callback ignored long_name altogether. + vars = self.data_variables + [ + _data_variable(mocker, "var1", long_name="y component of wind"), + ] + result = [callback(var) for var in vars] + _shared_utils.assert_array_equal(result, [True, True, True, True, True, False]) def test_name_constraint_var_name(self): constr = iris.NameConstraint(var_name="var1") @@ -101,7 +112,7 @@ def test_name_constraint_standard_name_var_name(self): constr = iris.NameConstraint(standard_name="x_wind", var_name="var1") callback = _translate_constraints_to_var_callback(constr) result = [callback(var) for var in self.data_variables] - _shared_utils.assert_array_equal(result, [True, False, False, True, False]) + _shared_utils.assert_array_equal(result, [True, False, True, True, True]) def test_name_constraint_standard_name_long_name_var_name(self): constr = iris.NameConstraint( @@ -111,7 +122,7 @@ def test_name_constraint_standard_name_long_name_var_name(self): ) callback = _translate_constraints_to_var_callback(constr) result = [callback(var) for var in self.data_variables] - _shared_utils.assert_array_equal(result, [False, False, False, True, False]) + _shared_utils.assert_array_equal(result, [True, False, True, True, True]) def test_name_constraint_with_stash(self): constr = iris.NameConstraint(standard_name="x_wind", STASH="m01s00i024") diff --git a/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py index 1451bfdae9..d5dd726066 100644 --- a/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py +++ b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver.py @@ -4,7 +4,6 @@ # See LICENSE in the root of the repository for full licensing details. """Unit tests for the :class:`iris.fileformats.netcdf.Saver` class.""" -import collections from contextlib import contextmanager from pathlib import Path from types import ModuleType @@ -32,6 +31,7 @@ from iris.cube import Cube from iris.fileformats.netcdf import Saver, _thread_safe_nc from iris.fileformats.netcdf import _bytecoding_datasets as ds_wrappers +from iris.fileformats.netcdf._dataset import NetCDFDatasetVariable from iris.tests import _shared_utils from iris.tests._shared_utils import assert_CDL import iris.tests.stock as stock @@ -217,22 +217,18 @@ def test_big_endian(self, request, tmp_path): def test_zlib(self, mocker): cube = self._simple_cube(">f4") - api = mocker.patch("iris.fileformats.netcdf.saver.bytecoding_datasets") - # Define mocked default fill values to prevent deprecation warning (#4374). - api.default_fillvals = collections.defaultdict(lambda: -99.0) + api = mocker.patch("iris.fileformats.netcdf._dataset._bytecoding_datasets") # Mock the apparent dtype of mocked variables, to avoid an error. - ref = api.DatasetWrapper.return_value - ref = ref.createVariable.return_value - ref.dtype = np.dtype(np.float32) + dataset = api.EncodedDataset.return_value + dataset.createVariable.return_value.dtype = np.dtype(np.float32) # NOTE: use compute=False as otherwise it gets in a pickle trying to construct # a fill-value report on a non-compliant variable in a non-file (!) with Saver("/dummy/path", "NETCDF4", compute=False) as saver: saver.write(cube, zlib=True) - dataset = api.EncodedDataset.return_value create_var_call = mocker.call( "air_pressure_anomaly", np.dtype("float32"), - ["dim0", "dim1"], + ("dim0", "dim1"), fill_value=None, shuffle=True, least_significant_digit=None, @@ -276,7 +272,7 @@ def test_compression(self, mocker, tmp_path): tgt, # Use 'wraps' to allow the patched methods to function as normal # - the patch object just acts as a 'spy' on its calls. - wraps=saver._dataset.createVariable, + wraps=saver._dataset.dataset.createVariable, ) saver.write(cube, **compression_kwargs) @@ -316,7 +312,7 @@ def test_non_compression__shape(self, mocker, tmp_path): tgt, # Use 'wraps' to allow the patched methods to function as normal # - the patch object just acts as a 'spy' on its calls. - wraps=saver._dataset.createVariable, + wraps=saver._dataset.dataset.createVariable, ) saver.write(cube, **compression_kwargs) @@ -356,7 +352,7 @@ def test_non_compression__dtype(self, mocker, tmp_path): tgt, # Use 'wraps' to allow the patched methods to function as normal # - the patch object just acts as a 'spy' on its calls. - wraps=saver._dataset.createVariable, + wraps=saver._dataset.dataset.createVariable, ) saver.write(cube, **compression_kwargs) @@ -506,24 +502,26 @@ def _check_bounds_setting(self, climatological=False): saver._ensure_valid_dtype.return_value = self.mocker.Mock( shape=coord.bounds.shape, dtype=coord.bounds.dtype ) - var = self.mocker.MagicMock(spec=ds_wrappers.EncodedVariable) + var = self.mocker.MagicMock(spec=NetCDFDatasetVariable) + var.attributes = {} + var.dimensions = ("time",) # Make the main call. Saver._create_cf_bounds(saver, coord, var, "time") - # Test the call of _setncattr in _create_cf_bounds. - setncattr_call = self.mocker.call( - property_name, boundsvar_name.encode(encoding="ascii") - ) - assert setncattr_call == var.setncattr.call_args + # Test the attribute written by _create_cf_bounds. The ASCII-to-bytes + # coercion now happens inside _NetCDFAttributes.__setitem__, and is + # tested there; this plain dict stands in for one, so the value + # arrives as given. + assert var.attributes[property_name] == boundsvar_name - # Test the call of createVariable in _create_cf_bounds. + # Test the call of create_variable in _create_cf_bounds. dataset = saver._dataset expected_dimensions = var.dimensions + ("bnds",) create_var_call = self.mocker.call( boundsvar_name, coord.bounds.dtype, expected_dimensions ) - assert create_var_call == dataset.createVariable.call_args + assert create_var_call == dataset.create_variable.call_args def test_set_bounds_default(self): self._check_bounds_setting(climatological=False) @@ -742,8 +740,10 @@ def assert_attribute(self, value): def check_attribute_compliance_call(self, value, file_type="NETCDF4"): self.set_attribute(value) with Saver("nonexistent test file", file_type) as saver: - # Get the Mock to work properly. - saver._dataset.file_format = file_type + # Get the Mock to work properly. The format is read from the + # netCDF dataset itself, through Saver's escape hatch; setting it + # on the NetCDFDataset would bind an attribute nothing consults. + saver._dataset.dataset.file_format = file_type saver.check_attribute_compliance(self.container, self.data_dtype) @@ -1050,12 +1050,19 @@ class NCMock(self.mocker.Mock): def setncattr(self, name, attr): setattr(self, name, attr) + def ncattrs(self): + # A variable the dataset wraps is asked for its attributes as + # it is wrapped; this one starts with none. + return [] + # Calls the actual NetCDF saver with appropriate mocking, returning # the grid variable that gets created. grid_variable = NCMock(name="NetCDFVariable") create_var_fn = self.mocker.Mock(side_effect=[grid_variable]) - dataset = self.mocker.Mock(variables=[], createVariable=create_var_fn) - variable = NCMock() + # 'variables' is a mapping, as netCDF4 presents it: the saver looks + # names up in it to avoid a grid-mapping variable name collision. + dataset = self.mocker.Mock(variables={}, createVariable=create_var_fn) + variable = NCMock(attributes={}) saver = Saver(dataset, "NETCDF4", compute=False) @@ -1063,7 +1070,13 @@ def setncattr(self, name, attr): saver._create_cf_grid_mapping(cube, variable) assert create_var_fn.call_count == 1 - assert variable.grid_mapping, grid_variable.grid_mapping_name + # The data variable refers to the grid variable by name. Under + # extended grid mapping the reference is followed by the coordinates + # it applies to, so compare the name alone. The grid variable's own + # name comes back as bytes, from the ASCII coercion that every + # attribute write goes through. + referenced_name = variable.attributes["grid_mapping"].split(":")[0] + assert referenced_name == grid_variable.grid_mapping_name.decode() return grid_variable def _variable_attributes(self, coord_system): diff --git a/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py new file mode 100644 index 0000000000..66b8cb3176 --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__dataset.py @@ -0,0 +1,104 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Tests for the dataset :class:`iris.fileformats.netcdf.saver.Saver` owns.""" + +import numpy as np +import pytest + +from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc +from iris.fileformats.netcdf._dataset import NetCDFDataset +from iris.fileformats.netcdf.saver import Saver + + +class TestOwnedDataset: + @pytest.fixture + def saver(self, tmp_path): + with Saver(tmp_path / "test.nc", "NETCDF4") as saver: + yield saver + + def test_dataset_is_a_netcdf_cfdataset(self, saver): + assert isinstance(saver._dataset, NetCDFDataset) + + def test_filepath_is_still_a_path(self, saver, tmp_path): + # Public API: Saver.filepath has always been a Path for a local file. + assert saver.filepath == (tmp_path / "test.nc").absolute() + + def test_the_write_lock_is_the_datasets_own(self, saver): + # Every variable of the dataset holds this same lock. A second lock + # would exclude nothing - see NetCDFDataset.write_lock. + assert saver.file_write_lock is saver._dataset.write_lock + + def test_format_reaches_the_file(self, tmp_path): + # Not through the context manager: __exit__ syncs, and a netCDF3 file + # with nothing written to it is still in define mode, where sync() + # raises. That is as true of the saver before this change as after. + saver = Saver(tmp_path / "classic.nc", "NETCDF3_CLASSIC") + try: + assert saver._dataset.dataset.file_format == "NETCDF3_CLASSIC" + finally: + saver._dataset.close() + + def test_exit_closes_an_owned_dataset(self, tmp_path): + with Saver(tmp_path / "test.nc", "NETCDF4") as saver: + pass + assert saver._dataset.closed + + +class TestBorrowedDataset: + @pytest.fixture + def raw(self, tmp_path): + dataset = threadsafe_nc.DatasetWrapper( + tmp_path / "user.nc", mode="w", format="NETCDF4" + ) + yield dataset + if dataset.isopen(): + dataset.close() + + def test_dataset_is_a_netcdf_cfdataset(self, raw): + with Saver(raw, "NETCDF4", compute=False) as saver: + assert isinstance(saver._dataset, NetCDFDataset) + + def test_exit_leaves_a_borrowed_dataset_open(self, raw): + with Saver(raw, "NETCDF4", compute=False) as saver: + pass + assert raw.isopen() + assert not saver._dataset.closed + + def test_complete_refuses_while_the_file_is_open(self, raw): + with Saver(raw, "NETCDF4", compute=False) as saver: + pass + with pytest.raises(ValueError, match="until its dataset is closed"): + saver.complete() + + def test_complete_sees_a_dataset_closed_behind_its_back(self, raw): + # The caller owns the dataset and closes it themselves, so the flag + # NetCDFDataset.close() sets is never set. complete() has to ask the + # file, not the flag. + with Saver(raw, "NETCDF4", compute=False) as saver: + pass + raw.close() + assert saver._dataset.closed + saver.complete() # must not raise + + +class TestWritingThroughTheDataset: + """The four verbs, through CFDataset rather than netCDF4.""" + + @pytest.fixture + def saver(self, tmp_path): + with Saver(tmp_path / "test.nc", "NETCDF4") as saver: + yield saver + + def test_create_dimension(self, saver): + saver._dataset.create_dimension("x", 3) + assert saver._dataset.dimensions["x"] == 3 + + def test_create_variable_returns_a_cf_dataset_variable(self, saver): + from iris.fileformats.netcdf._dataset import NetCDFDatasetVariable + + saver._dataset.create_dimension("x", 3) + variable = saver._dataset.create_variable("a", np.dtype("f4"), ("x",)) + assert isinstance(variable, NetCDFDatasetVariable) + assert saver._dataset.variables["a"] is variable diff --git a/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py index 3f1c38661b..167ef0a8a8 100644 --- a/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py +++ b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__lazy_stream_data.py @@ -16,8 +16,7 @@ import numpy as np import pytest -import iris.fileformats.netcdf._bytecoding_datasets as bytecoding_datasets -import iris.fileformats.netcdf._thread_safe_nc as threadsafe_nc +import iris.fileformats.netcdf._dataset as dataset_module from iris.fileformats.netcdf.saver import Saver @@ -61,18 +60,15 @@ def saver(compute, data_form, tmp_path) -> Saver: @staticmethod def mock_var(shape, with_data_array, mocker, dtype=np.dtype(np.float32)): # Create a test cf_var object. - # N.B. using 'spec=' so we can control whether it has a '_data_array' property. - if with_data_array: - extra_properties = {"_data_array": mocker.sentinel.initial_data_array} - else: - extra_properties = {} + # 'is_emulated' is now a declared property rather than the presence of + # a '_data_array' member, so it can simply be set. mock_cfvar = mocker.MagicMock( - spec=threadsafe_nc.VariableWrapper, + spec=dataset_module.NetCDFDatasetVariable, shape=tuple(shape), dtype=dtype, - _contained_instance=mocker.Mock(dtype="f4"), - **extra_properties, + is_emulated=with_data_array, ) + mock_cfvar.write_handle.return_value = mocker.sentinel.write_handle # Give the mock cf-var a name property, as required by '_lazy_stream_data'. # This *can't* be an extra kwarg to MagicMock __init__, since that already # defines a specific 'name' kwarg, with a different purpose. @@ -108,12 +104,14 @@ def test_data_save(self, compute, data_form, mocker, tmp_path): if data_form == "lazydata": result_data, result_writer = saver._delayed_writes[0] assert result_data is data - assert isinstance(result_writer, threadsafe_nc.NetCDFWriteProxy) + # What kind of handle it is is the dataset's business, and is + # tested in test_NetCDFDataset__write.py. + assert result_writer is mocker.sentinel.write_handle elif data_form == "realdata": cf_var.__setitem__.assert_called_once_with(slice(None), data) else: assert data_form == "emulateddata" - assert cf_var._data_array is data + assert cf_var.emulated_data_array is data @pytest.mark.parametrize("is_realdata", [True, False], ids=["realdata", "lazydata"]) @pytest.mark.parametrize("is_string", [True, False], ids=["string", "numeric"]) @@ -135,13 +133,21 @@ def test_data_save_emulated_data_array( ) if is_string: - contained = cf_var._contained_instance - contained.name = "" - contained.dtype = np.dtype("S1") - contained.dimensions = ("x", "strlen") - mock_group = mocker.Mock() - mock_group.dimensions = {"strlen": mocker.Mock(size=5)} - contained.group.return_value = mock_group + # The encoding is described by the file variable beneath the + # emulation, which is 'char' where the data written is strings. + unencoded = mocker.Mock( + dtype=np.dtype("S1"), + dimensions=("x", "strlen"), + # None is what an absent _Encoding attribute yields: the + # default encoding, and no "unsupported encoding" warning. + _Encoding=None, + ) + # .name is a Mock constructor keyword, so it has to be set after. + unencoded.name = "" + unencoded.group.return_value = mocker.Mock( + dimensions={"strlen": mocker.Mock(size=5)} + ) + cf_var.unencoded_variable = unencoded ifnbd = "iris.fileformats.netcdf._bytecoding_datasets." mock_encode = mocker.patch(ifnbd + "encode_stringarray_as_bytearray") @@ -154,16 +160,16 @@ def test_data_save_emulated_data_array( assert len(saver._nczarr_writes) == 0 if not is_string: - assert cf_var._data_array is data + assert cf_var.emulated_data_array is data else: if is_realdata: assert mock_encode.call_count == 1 - assert cf_var._data_array is mock_encode.return_value + assert cf_var.emulated_data_array is mock_encode.return_value call_args = mock_encode.call_args_list[0][0] assert call_args[0] is data else: assert mock_mapblocks.call_count == 1 - assert cf_var._data_array is mock_mapblocks.return_value + assert cf_var.emulated_data_array is mock_mapblocks.return_value call_args = mock_mapblocks.call_args_list[0][0] assert call_args[0] is mock_encode assert call_args[1] is data @@ -201,5 +207,7 @@ def test_exit_flushes_nczarr_writes(self, data_form, mocker, tmp_path): saver.__exit__(None, None, None) store_patch.assert_called_once_with([source], [target]) - saver._dataset.sync.assert_called_once_with() - saver._dataset.close.assert_called_once_with() + # Saver._dataset is a NetCDFDataset now, so the mock that records the + # calls is the netCDF dataset it opened, one level down. + saver._dataset.dataset.sync.assert_called_once_with() + saver._dataset.dataset.close.assert_called_once_with() diff --git a/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py new file mode 100644 index 0000000000..ef0bb7c93a --- /dev/null +++ b/lib/iris/tests/unit/fileformats/netcdf/saver/test_Saver__user_dataset.py @@ -0,0 +1,273 @@ +# Copyright Iris contributors +# +# This file is part of Iris and is released under the BSD license. +# See LICENSE in the root of the repository for full licensing details. +"""Tests for saving into a dataset the caller opened, or is only pretending to. + +``iris.fileformats.netcdf.save(cube, dataset, compute=False)`` accepts an +open dataset in place of a path. Two kinds of caller do this: + +* one holding a real, open netCDF4 dataset, who wants Iris to add to it; +* one holding an object that merely *looks* like a netCDF4 dataset, and + collects what Iris writes rather than storing it - the "Xarray bridge", + https://github.com/SciTools/iris/issues/4994, which is how + https://github.com/pp-mo/ncdata translates between Iris and Xarray. + +The second is the reason ``Saver`` may not assume its dataset is netCDF4, +and it is entirely untested elsewhere. + +Saving through a ``CFDataset`` asks an emulating object for three members +that the saver never used to reach for. All three are public netCDF4 API, +and all three are now required of an emulator: + +* ``Dimension.size`` - the CF layer reads dimension lengths through it. + ``__len__`` is not an alternative, as ``_EmulatedDimension`` explains. +* ``Dataset.ncattrs()`` - read once, as the dataset is wrapped. +* ``Variable.ncattrs()`` - read once per variable, as each is wrapped. + +They are marked below where the emulators define them. This is an accepted +consequence of the change, not an oversight: there is deliberately no +fallback for an emulator that lacks them. + +""" + +import dask +import dask.array as da +import numpy as np +import pytest + +import iris +from iris.coords import DimCoord +from iris.cube import Cube +import iris.fileformats.netcdf as inetcdf +from iris.fileformats.netcdf import _thread_safe_nc as threadsafe_nc + + +@pytest.fixture(autouse=True) +def split_attrs(): + # Avoid the legacy-attribute deprecation warning; irrelevant here. + with iris.FUTURE.context(save_split_attrs=True): + yield + + +@pytest.fixture +def cube(): + cube = Cube( + np.arange(6.0).reshape(2, 3), + standard_name="air_temperature", + units="K", + var_name="air", + ) + cube.add_dim_coord( + DimCoord(np.arange(2.0), standard_name="latitude", units="degrees"), 0 + ) + cube.add_dim_coord( + DimCoord(np.arange(3.0), standard_name="longitude", units="degrees"), 1 + ) + return cube + + +class TestRealDataset: + """A dataset the caller opened, and closes themselves.""" + + def test_save_into_an_open_dataset(self, cube, tmp_path): + path = tmp_path / "user.nc" + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + delayed = inetcdf.save(cube, dataset, compute=False) + # The caller owns the dataset, so the caller closes it - and the + # delayed writes only resolve once they have. + dataset.close() + dask.compute(delayed) + + result = iris.load_cube(path) + assert result.standard_name == "air_temperature" + assert result.units == "K" + assert np.array_equal(result.data, cube.data) + assert [coord.name() for coord in result.coords()] == [ + "latitude", + "longitude", + ] + + def test_saver_does_not_close_what_it_was_given(self, cube, tmp_path): + path = tmp_path / "user.nc" + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + inetcdf.save(cube, dataset, compute=False) + assert dataset.isopen() + dataset.close() + + def test_deferred_string_writes_are_encoded(self, tmp_path): + """A deferred unicode write into a borrowed dataset must reach the file intact. + + ``NetCDFDataset.from_existing`` only wraps a dataset that lacks + ``THREAD_SAFE_FLAG``, so a borrowed ``DatasetWrapper`` keeps plain, + unencoded variables - and a write handle taken from one of those must + encode all the same, because writes have no selectable string + encoding. Asserting on the *file*, not on the handle's class: a handle + of the right class that still wrote the wrong bytes would pass a + type assertion. Unfixed, ``["abc", "def"]`` read back as + ``["aaa", "ddd"]``. + """ + path = tmp_path / "strings.nc" + strings = np.array(["abc", "def"], dtype="U3") + # Lazy, and chunked, so that the save really is deferred to da.store. + cube = Cube(da.from_array(strings, chunks=1), long_name="strings") + + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + delayed = inetcdf.save(cube, dataset, compute=False) + dataset.close() + dask.compute(delayed) + + result = iris.load_cube(path) + assert result.dtype.kind == "U" + np.testing.assert_array_equal(result.data, strings) + + def test_compute_true_is_refused(self, cube, tmp_path): + path = tmp_path / "user.nc" + dataset = threadsafe_nc.DatasetWrapper(path, mode="w", format="NETCDF4") + with pytest.raises(ValueError, match="Cannot save to a user-provided dataset"): + inetcdf.save(cube, dataset, compute=True) + dataset.close() + + +class _EmulatedDimension: + def __init__(self, size): + self._size = size + # NEWLY REQUIRED OF AN EMULATOR, 1 of 3 - see the module docstring. + # netCDF4.Dimension.size, which is how the length is read back. len() + # is not an alternative: the emulator is put inside a _thread_safe_nc + # wrapper, whose __getattr__ forwards named members but is never + # consulted for the len() protocol. + self.size = size + + def __len__(self): + return self._size + + def isunlimited(self): + return self._size is None + + +class _EmulatedVariable: + """The least a netCDF4.Variable emulator can be and still be written to. + + ``_data_array`` is the whole point: an emulator receives its data by + having this attribute set, never by ``__setitem__``. ``__setitem__`` + raises here so that a regression shows up as a failure rather than as + data quietly going nowhere. + + """ + + def __init__(self, name, datatype, dimensions, shape): + self.name = name + self.datatype = np.dtype(datatype) + self.dtype = self.datatype + self.dimensions = tuple(dimensions) + self.shape = shape + self.size = int(np.prod(shape)) if shape else 1 + self._data_array = None + self._attrs = {} + + def setncattr(self, name, value): + self._attrs[name] = value + + def getncattr(self, name): + return self._attrs[name] + + def ncattrs(self): + # NEWLY REQUIRED OF AN EMULATOR, 2 of 3 - see the module docstring. + # Read once, as the variable is wrapped for the attribute mapping. + return list(self._attrs) + + def chunking(self): + return "contiguous" + + def __setitem__(self, keys, values): + raise AssertionError("An emulated variable must receive _data_array.") + + +class _EmulatedDataset: + """A netCDF4.Dataset emulator, in the shape ncdata presents. + + Deliberately *not* a ``_thread_safe_nc`` wrapper and deliberately without + ``THREAD_SAFE_FLAG``, so that the wrapping branch of + ``NetCDFDataset.from_existing`` is the one under test. + + """ + + def __init__(self): + self.variables = {} + self.dimensions = {} + self.file_format = "NETCDF4" + self._attrs = {} + self._open = True + + def createDimension(self, name, size): + self.dimensions[name] = _EmulatedDimension(size) + return self.dimensions[name] + + def createVariable(self, name, datatype, dimensions=(), **kwargs): + shape = tuple(len(self.dimensions[name_]) for name_ in dimensions) + variable = _EmulatedVariable(name, datatype, dimensions, shape) + self.variables[name] = variable + return variable + + def setncattr(self, name, value): + self._attrs[name] = value + + def getncattr(self, name): + return self._attrs[name] + + def ncattrs(self): + # NEWLY REQUIRED OF AN EMULATOR, 3 of 3 - see the module docstring. + # Read once, as the dataset is wrapped for the attribute mapping. + return list(self._attrs) + + def sync(self): + pass + + def close(self): + self._open = False + + def isopen(self): + return self._open + + def filepath(self): + return "" + + def set_auto_chartostring(self, onoff): + pass + + +class TestEmulatedDataset: + """An object that only looks like a dataset - the Xarray bridge.""" + + @pytest.fixture + def written(self, cube): + dataset = _EmulatedDataset() + inetcdf.save(cube, dataset, compute=False) + return dataset + + def test_variables_are_created(self, written): + assert sorted(written.variables) == ["air", "latitude", "longitude"] + + def test_dimensions_are_created(self, written): + assert {name: len(dim) for name, dim in written.dimensions.items()} == { + "latitude": 2, + "longitude": 3, + } + + def test_attributes_reach_the_emulator_as_bytes(self, written): + # _bytes_if_ascii: an ASCII string attribute is offered as bytes, so + # that netCDF4 gives it type NC_CHAR. An emulator sees the same. + assert written.variables["air"]._attrs == { + "standard_name": b"air_temperature", + "units": b"K", + } + assert written._attrs == {"Conventions": b"CF-1.7"} + + def test_data_arrives_as_a_data_array(self, written, cube): + # Not via __setitem__, which the emulated variable refuses. + assert np.array_equal(written.variables["air"]._data_array, cube.data) + assert np.array_equal(written.variables["latitude"]._data_array, np.arange(2.0)) + + def test_dimensions_are_recorded_on_the_variable(self, written): + assert written.variables["air"].dimensions == ("latitude", "longitude")