diff --git a/docs/guide.md b/docs/guide.md index 456ad91..96ee3e0 100644 --- a/docs/guide.md +++ b/docs/guide.md @@ -67,11 +67,12 @@ types. Configuration sequences are again merged into one. import os from zappend.api import zappend -zappend(os.listdir("inputs"), - config=["configs/base.yaml", - "configs/mycube.yaml"], - target_dir="outputs/mycube.zarr", - dry_run=True) +zappend( + os.listdir("inputs"), + config=["configs/base.yaml", "configs/mycube.yaml"], + target_dir="outputs/mycube.zarr", + dry_run=True, +) ``` The remainder of this guide explains the how to use the various `zappend` @@ -463,8 +464,8 @@ You can compute `scale_factor` and `add_offset` from given data range in physica according to ```python - add_offset = memory_value_min - scale_factor = (memory_value_max - memory_value_min) / (2 ** num_bits - 1) +add_offset = memory_value_min +scale_factor = (memory_value_max - memory_value_min) / (2**num_bits - 1) ``` with `num_bits` being the number of bits for the integer type to be used. @@ -706,6 +707,7 @@ an `xarray.Dataset`: ```python import xarray as xr + # Slice source argument `path` is just an example. def get_dataset(path: str) -> xr.Dataset: # Provide dataset here. No matter how, e.g.: @@ -721,6 +723,7 @@ you can turn your slice source function into a from contextlib import contextmanager import xarray as xr + # Slice source argument `path` is just an example. @contextmanager def get_dataset(path: str) -> xr.Dataset: @@ -747,6 +750,7 @@ You can also implement your slice source as a class derived from the abstract import xarray as xr from zappend.api import SliceSource + class MySliceSource(SliceSource): # Slice source argument `path` is just an example. def __init__(self, path: str): @@ -781,9 +785,11 @@ qualified name of the slice source function or class: If you use the `zappend` function, you can pass the function or class directly: ```python -zappend(["slice-1.nc", "slice-2.nc", "slice-3.nc"], - target_dir="target.zarr", - slice_source=MySliceSource) +zappend( + ["slice-1.nc", "slice-2.nc", "slice-3.nc"], + target_dir="target.zarr", + slice_source=MySliceSource, +) ``` If the slice source setting is used, each slice item passed to `zappend` is passed as @@ -835,6 +841,7 @@ from zappend.api import Context from zappend.api import SliceSource from zappend.api import zappend + class MySliceSource(SliceSource): def __init__(self, ctx: Context, slice_path: str): self.quantiles = ctx.config.extra.get("quantiles", [0.5]) @@ -849,25 +856,28 @@ class MySliceSource(SliceSource): if self.ds is not None: self.ds.close() - def get_agg_slice(self, slice_ds: xr.Dataset) -> xr.Dataset: + def get_agg_slice(self, slice_ds: xr.Dataset) -> xr.Dataset: agg_slice_ds = slice_ds.quantile(self.quantiles, dim="time") # Re-introduce time dimension of size one agg_slice_ds = agg_slice_ds.expand_dims("time", axis=0) agg_slice_ds.coords["time"] = self.get_mean_time(slice_ds) - return agg_slice_ds + return agg_slice_ds @classmethod def get_mean_time(cls, slice_ds: xr.Dataset) -> xr.DataArray: time = slice_ds.time t0 = time[0] dt = time[-1] - t0 - return xr.DataArray(np.array([t0 + dt / 2], - dtype=slice_ds.time.dtype), - dims="time") - -zappend(["slice-1.nc", "slice-2.nc", "slice-3.nc"], - target_dir="target.zarr", - slice_source=MySliceSource) + return xr.DataArray( + np.array([t0 + dt / 2], dtype=slice_ds.time.dtype), dims="time" + ) + + +zappend( + ["slice-1.nc", "slice-2.nc", "slice-3.nc"], + target_dir="target.zarr", + slice_source=MySliceSource, +) ``` ## Profiling diff --git a/docs/hooks/download_fonts.py b/docs/hooks/download_fonts.py index 3916a9c..2aaab58 100644 --- a/docs/hooks/download_fonts.py +++ b/docs/hooks/download_fonts.py @@ -1,4 +1,4 @@ -# Copyright © 2024-2026 Brockmann Consult and contributors +# Copyright © 2024-2026 Brockmann Consult and contributors # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. diff --git a/docs/howdoi.md b/docs/howdoi.md index 7d7e13f..a608fc8 100644 --- a/docs/howdoi.md +++ b/docs/howdoi.md @@ -16,10 +16,11 @@ import rioxarray as rxr import xarray as xr from zappend.api import zappend + def get_dataset_from_geotiff(tiff_path): ds = rxr.open_rasterio(tiff_path) # Add missing time dimension - slice_time = get_slice_time(tiff_path) + slice_time = get_slice_time(tiff_path) slice_ds = ds.expand_dims("time", axis=0) slice_ds.coords["time"] = xr.Dataset(np.array([slice_time]), dims="time") try: @@ -27,9 +28,12 @@ def get_dataset_from_geotiff(tiff_path): finally: ds.close() -zappend(sorted(glob.glob("inputs/*.tif")), - slice_source=get_dataset_from_geotiff, - target_dir="output/tif-cube.zarr") + +zappend( + sorted(glob.glob("inputs/*.tif")), + slice_source=get_dataset_from_geotiff, + target_dir="output/tif-cube.zarr", +) ``` In the example above, function `get_slice_time()` returns the time label diff --git a/docs/start.md b/docs/start.md index 59c11c2..3f197e5 100644 --- a/docs/start.md +++ b/docs/start.md @@ -66,17 +66,21 @@ Process list of slices stored in S3 [configuration](config.md) in `config`: ```python from zappend.api import zappend -config = { +config = { "target_dir": "target.zarr", "slice_storage_options": { - "key": "...", - "secret": "...", - } + "key": "...", + "secret": "...", + }, } -zappend((f"s3:/mybucket/data/{name}" - for name in ["slice-1.nc", "slice-2.nc", "slice-3.nc"]), - config=config) +zappend( + ( + f"s3:/mybucket/data/{name}" + for name in ["slice-1.nc", "slice-2.nc", "slice-3.nc"] + ), + config=config, +) ``` Slice items can also be arguments passed to your custom _slice source_, @@ -91,9 +95,12 @@ def get_dataset(path: str): ds = xr.open_dataset(path) return ds.drop_vars(["ndvi_min", "ndvi_max"]) -zappend(["slice-1.nc", "slice-2.nc", "slice-3.nc"], - slice_source=get_dataset, - target_dir="target.zarr") + +zappend( + ["slice-1.nc", "slice-2.nc", "slice-3.nc"], + slice_source=get_dataset, + target_dir="target.zarr", +) ``` For the details, please refer to the section [_Slice Sources_](guide.md#slice-sources) in the diff --git a/examples/zappend-demo.ipynb b/examples/zappend-demo.ipynb index 0b8918b..152b614 100644 --- a/examples/zappend-demo.ipynb +++ b/examples/zappend-demo.ipynb @@ -593,7 +593,9 @@ } ], "source": [ - "slice_ds.SPM.isel(time=0).sel(lon=slice(9.2, 12), lat=slice(58.7, 56)).plot.imshow(vmax=1.5)" + "slice_ds.SPM.isel(time=0).sel(lon=slice(9.2, 12), lat=slice(58.7, 56)).plot.imshow(\n", + " vmax=1.5\n", + ")" ] }, { @@ -1768,7 +1770,9 @@ } ], "source": [ - "target_ds.SPM.mean(\"time\").sel(lon=slice(9.2, 12), lat=slice(58.7, 56)).plot.imshow(vmax=1.5)" + "target_ds.SPM.mean(\"time\").sel(lon=slice(9.2, 12), lat=slice(58.7, 56)).plot.imshow(\n", + " vmax=1.5\n", + ")" ] }, { @@ -1799,7 +1803,9 @@ } ], "source": [ - "target_ds.SPM_QI.max(\"time\").sel(lon=slice(9.2, 12), lat=slice(58.7, 56)).plot.imshow(vmax=100.0)" + "target_ds.SPM_QI.max(\"time\").sel(lon=slice(9.2, 12), lat=slice(58.7, 56)).plot.imshow(\n", + " vmax=100.0\n", + ")" ] }, { @@ -2959,7 +2965,7 @@ "outputs": [], "source": [ "# Uncomment to show configuration refernce\n", - "#Markdown(get_config_schema(format=\"md\"))" + "# Markdown(get_config_schema(format=\"md\"))" ] }, { @@ -2970,44 +2976,31 @@ "outputs": [], "source": [ "config = {\n", - " \"target_dir\": target_path, \n", + " \"target_dir\": target_path,\n", " \"variables\": {\n", " # We want the time coordinate variable to use a larger chunk size\n", " # than the default (= 1 here)\n", - " \"time\": {\n", - " \"encoding\": {\n", - " \"chunks\": [100]\n", - " }\n", - " }\n", + " \"time\": {\"encoding\": {\"chunks\": [100]}}\n", " },\n", " # Log to the console.\n", " # Note you could also configure the log output for dask here.\n", " \"logging\": {\n", " \"version\": 1,\n", " \"formatters\": {\n", - " \"normal\": {\n", - " \"format\": \"%(asctime)s %(levelname)s %(message)s\",\n", - " \"style\": \"%\"\n", - " }\n", + " \"normal\": {\"format\": \"%(asctime)s %(levelname)s %(message)s\", \"style\": \"%\"}\n", " },\n", " \"handlers\": {\n", - " \"console\": {\n", - " \"class\": \"logging.StreamHandler\",\n", - " \"formatter\": \"normal\"\n", - " }\n", + " \"console\": {\"class\": \"logging.StreamHandler\", \"formatter\": \"normal\"}\n", " },\n", " \"loggers\": {\n", - " \"zappend\": {\n", - " \"level\": \"INFO\",\n", - " \"handlers\": [\"console\"]\n", - " }, \n", + " \"zappend\": {\"level\": \"INFO\", \"handlers\": [\"console\"]},\n", " \"notebook\": {\n", " # Will use this one later below\n", " \"level\": \"INFO\",\n", - " \"handlers\": [\"console\"]\n", - " }\n", - " }\n", - " }\n", + " \"handlers\": [\"console\"],\n", + " },\n", + " },\n", + " },\n", "}" ] }, @@ -3029,6 +3022,7 @@ "outputs": [], "source": [ "import yaml\n", + "\n", "with open(config_path, mode=\"w\") as f:\n", " yaml.dump(config, f)" ] @@ -4189,6 +4183,7 @@ "outputs": [], "source": [ "from logging import getLogger\n", + "\n", "LOG = getLogger(\"notebook\")" ] }, @@ -4209,11 +4204,13 @@ "source": [ "log = getLogger(\"notebook\")\n", "\n", + "\n", "def process_slice(slice_path: str) -> xr.Dataset:\n", " LOG.info(f\"Processing slice {slice_path}\")\n", - " slice_ds = xr.open_dataset(slice_path) \n", + " slice_ds = xr.open_dataset(slice_path)\n", " return slice_ds.drop_vars([\"SPM_QI\", \"TUR_QI\", \"crs\"])\n", "\n", + "\n", "slice_generator = (process_slice(slice_path) for slice_path in slice_paths)" ] }, @@ -4999,16 +4996,17 @@ "source": [ "from zappend.api import SliceSource\n", "\n", + "\n", "class MySliceSource(SliceSource):\n", " def __init__(self, slice_path):\n", " self.slice_path = slice_path\n", " self.slice_ds = None\n", - " \n", - " def get_dataset(self) -> xr.Dataset: \n", + "\n", + " def get_dataset(self) -> xr.Dataset:\n", " LOG.info(f\"Processing slice {self.slice_path}\")\n", " self.slice_ds = xr.open_dataset(self.slice_path)\n", " return self.slice_ds.drop_vars([\"SPM_QI\", \"TUR_QI\", \"crs\"])\n", - " \n", + "\n", " def dispose(self):\n", " self.slice_ds.close()\n", " self.slice_ds = None\n", @@ -5789,31 +5787,34 @@ "source": [ "from zappend.api import SliceSource\n", "\n", + "\n", "class MyGrottySliceSource(SliceSource):\n", " def __init__(self, slice_path):\n", " self.slice_path = slice_path\n", " self.slice_ds = None\n", " self.is_bad = \"20230605\" in slice_path\n", " self.num_chunks_seen = 0\n", - " \n", + "\n", " def half_spm(self, spm):\n", " if self.is_bad:\n", " self.num_chunks_seen += 1\n", " if self.num_chunks_seen == 12:\n", " raise ValueError(\"Bad chunk detected!\")\n", " return 0.5 * spm\n", - " \n", - " def get_dataset(self) -> xr.Dataset: \n", + "\n", + " def get_dataset(self) -> xr.Dataset:\n", " LOG.info(f\"Processing slice {self.slice_path}\")\n", - " slice_ds = xr.open_dataset(self.slice_path) \n", + " slice_ds = xr.open_dataset(self.slice_path)\n", " slice_ds = slice_ds.drop_vars([\"SPM_QI\", \"TUR_QI\", \"crs\"])\n", - " slice_ds = slice_ds.chunk(dict(lon=2000, lat=1000))\n", + " slice_ds = slice_ds.chunk({\"lon\": 2000, \"lat\": 1000})\n", " for v in slice_ds.data_vars.values():\n", - " v.encoding = {} \n", - " slice_ds[\"SPM_05\"] = slice_ds[\"SPM\"].map_blocks(self.half_spm, template=slice_ds[\"SPM\"]) \n", + " v.encoding = {}\n", + " slice_ds[\"SPM_05\"] = slice_ds[\"SPM\"].map_blocks(\n", + " self.half_spm, template=slice_ds[\"SPM\"]\n", + " )\n", " self.slice_ds = slice_ds\n", " return slice_ds\n", - " \n", + "\n", " def dispose(self):\n", " self.slice_ds.close()\n", " self.slice_ds = None\n", diff --git a/tests/config/test_attrs.py b/tests/config/test_attrs.py index b46124c..8104783 100644 --- a/tests/config/test_attrs.py +++ b/tests/config/test_attrs.py @@ -163,20 +163,20 @@ def test_scalar_result(self): time = datetime.date.fromisoformat("2024-01-02") self.assertEqual( "2024-01-02", - eval_expr("time", dict(time=time)), + eval_expr("time", {"time": time}), ) time = datetime.datetime.fromisoformat("2024-01-02T10:20:30") self.assertEqual( "2024-01-02T10:20:30", - eval_expr("time", dict(time=time)), + eval_expr("time", {"time": time}), ) with pytest.raises( ValueError, match="cannot serialize value of type " ): - eval_expr("obj", dict(obj=object())) + eval_expr("obj", {"obj": object()}) def test_dict_result(self): - self.assertEqual({}, eval_expr("d", dict(d={}))) + self.assertEqual({}, eval_expr("d", {"d": {}})) self.assertEqual( { "b": True, @@ -189,8 +189,8 @@ def test_dict_result(self): }, eval_expr( "d", - dict( - d={ + { + "d": { "b": True, "i": 13, "t": (1, "B", {}), @@ -199,7 +199,7 @@ def test_dict_result(self): "np_a": np.array([0.1, 0.2]), "xr_a": xr.DataArray(np.array([0.3, 0.4])), } - ), + }, ), ) @@ -218,7 +218,7 @@ def test_array_1d_result(self): " of type , dtype=dtype\\('O'\\)" ), ): - eval_expr("a", dict(a=xr.DataArray([object(), object()]))) + eval_expr("a", {"a": xr.DataArray([object(), object()])}) def test_array_2d_result(self): self.assert_array_ok([[3, 4], [5, 6]]) @@ -227,33 +227,33 @@ def assert_array_ok(self, a: list, dtype=None): # Test list self.assertEqual( a, - eval_expr("a", dict(a=a)), + eval_expr("a", {"a": a}), ) self.assertEqual( a[0], - eval_expr("a[0]", dict(a=a)), + eval_expr("a[0]", {"a": a}), ) # Test numpy.ndarray np_a = np.array(a, dtype=dtype) if dtype is not None else np.array(a) self.assertEqual( a, - eval_expr("a", dict(a=np_a)), + eval_expr("a", {"a": np_a}), ) self.assertEqual( a[0], - eval_expr("a[0]", dict(a=np_a)), + eval_expr("a[0]", {"a": np_a}), ) # Test xarray-DataArray xr_a = xr.DataArray(np_a) self.assertEqual( a, - eval_expr("a", dict(a=xr_a)), + eval_expr("a", {"a": xr_a}), ) self.assertEqual( a[0], - eval_expr("a[0]", dict(a=xr_a)), + eval_expr("a[0]", {"a": xr_a}), ) diff --git a/tests/config/test_normalize.py b/tests/config/test_normalize.py index e9ea2bb..193ce80 100644 --- a/tests/config/test_normalize.py +++ b/tests/config/test_normalize.py @@ -177,7 +177,7 @@ def test_normalize_invalid(self): with pytest.raises(TypeError): normalize_config(True) with pytest.raises(TypeError): - normalize_config(bytes()) + normalize_config(b"") def test_merge_config(self): self.assertEqual({}, merge_configs()) diff --git a/tests/contrib/test_levels.py b/tests/contrib/test_levels.py index de24725..7c58639 100644 --- a/tests/contrib/test_levels.py +++ b/tests/contrib/test_levels.py @@ -23,7 +23,7 @@ class GetVariablesConfigTest(unittest.TestCase): def test_no_variables_given(self): dataset = make_test_dataset() - variables = get_variables_config(dataset, dict(x=512, y=256, time=1)) + variables = get_variables_config(dataset, {"x": 512, "y": 256, "time": 1}) self.assertEqual( { "x": {"dims": ["x"], "encoding": {"chunks": None}}, @@ -45,7 +45,7 @@ def test_variables_given(self): dataset = make_test_dataset() variables = get_variables_config( dataset, - dict(x=512, y=256, time=1), + {"x": 512, "y": 256, "time": 1}, variables={ "time": {"encoding": {"chunks": [3]}}, "chl": {"encoding": {"chunks": [3, 100, 100]}}, @@ -116,20 +116,22 @@ def test_config_given(self): ) def test_dry_run_and_use_saved_levels_given(self): - with pytest.raises( - FileNotFoundError, - match="Target parent directory does not exist: /target.levels", - ): - with pytest.warns( + with ( + pytest.raises( + FileNotFoundError, + match="Target parent directory does not exist: /target.levels", + ), + pytest.warns( UserWarning, match="'use_saved_levels' argument is not applicable if dry_run=True", - ): - write_levels( - source_path=source_path, - target_path=target_path, - dry_run=True, - use_saved_levels=True, - ) + ), + ): + write_levels( + source_path=source_path, + target_path=target_path, + dry_run=True, + use_saved_levels=True, + ) def test_source_path_and_source_ds_not_given(self): with pytest.raises( diff --git a/tests/fsutil/test_fileobj.py b/tests/fsutil/test_fileobj.py index 6bb7a8e..5a6a965 100644 --- a/tests/fsutil/test_fileobj.py +++ b/tests/fsutil/test_fileobj.py @@ -17,9 +17,7 @@ def test_str(self): self.assertEqual("memory://test.zarr", str(FileObj("memory://test.zarr"))) self.assertEqual( "memory://test.zarr", - str( - FileObj("memory://test.zarr", storage_options=dict(asynchronous=False)) - ), + str(FileObj("memory://test.zarr", storage_options={"asynchronous": False})), ) def test_repr(self): @@ -29,7 +27,7 @@ def test_repr(self): self.assertEqual( "FileObj('memory://test.zarr', storage_options={'asynchronous': False})", repr( - FileObj("memory://test.zarr", storage_options=dict(asynchronous=False)) + FileObj("memory://test.zarr", storage_options={"asynchronous": False}) ), ) @@ -63,8 +61,8 @@ def test_hash(self): FileObj("s3://test.zarr", storage_options={"anon": False}), ] self.assertEqual(4, len(set(files))) - self.assertEqual(4, len(set(hash(f) for f in files))) - self.assertEqual(list(hash(f) for f in files), list(hash(f) for f in files)) + self.assertEqual(4, len({hash(f) for f in files})) + self.assertEqual([hash(f) for f in files], [hash(f) for f in files]) def test_memory_protocol(self): zarr_dir = FileObj("memory://test.zarr") diff --git a/tests/fsutil/test_transaction.py b/tests/fsutil/test_transaction.py index 753b412..61ee45c 100644 --- a/tests/fsutil/test_transaction.py +++ b/tests/fsutil/test_transaction.py @@ -3,7 +3,7 @@ # https://opensource.org/licenses/MIT. import unittest -from typing import Callable +from collections.abc import Callable import pytest @@ -148,24 +148,28 @@ def test_it_raises_on_nested_transaction(self): test_root.mkdir() rollback_dir = FileObj("memory://rollback") transaction = Transaction(test_root, rollback_dir) - with transaction: - with pytest.raises( + with ( + transaction, + pytest.raises( ValueError, match="Transaction instance cannot be" " used with nested 'with' statements", - ): - with transaction: - pass + ), + transaction, + ): + pass # noinspection PyMethodMayBeStatic def test_it_raises_on_locked_target(self): test_root = FileObj("memory://test") test_root.mkdir() rollback_dir = FileObj("memory://rollback") - with Transaction(test_root, rollback_dir): - with pytest.raises(OSError, match="Target is locked: memory://test.lock"): - with Transaction(test_root, rollback_dir): - pass + with ( + Transaction(test_root, rollback_dir), + pytest.raises(OSError, match="Target is locked: memory://test.lock"), + Transaction(test_root, rollback_dir), + ): + pass # noinspection PyMethodMayBeStatic def test_it_raises_if_not_used_with_with(self): @@ -215,52 +219,64 @@ def test_it_raises_on_illegal_callback_calls(self): test_root = FileObj("memory://test") test_root.mkdir() rollback_dir = FileObj("memory://rollback") - with pytest.raises( - TypeError, - match="Transaction._add_rollback_action\\(\\)" - " missing 3 required positional arguments:" - " 'action', 'path', and 'data'", + with ( + pytest.raises( + TypeError, + match="Transaction._add_rollback_action\\(\\)" + " missing 3 required positional arguments:" + " 'action', 'path', and 'data'", + ), + Transaction(test_root, rollback_dir) as callback, ): - with Transaction(test_root, rollback_dir) as callback: - callback() - - with pytest.raises( - TypeError, - match="Type of 'action' argument must be" - " , but was ", + callback() + + with ( + pytest.raises( + TypeError, + match="Type of 'action' argument must be" + " , but was ", + ), + Transaction(test_root, rollback_dir) as callback, ): - with Transaction(test_root, rollback_dir) as callback: - callback(42, "I/am/the/path", b"I/m/the/data") - - with pytest.raises( - TypeError, - match="Type of 'path' argument must be" - " , but was ", + callback(42, "I/am/the/path", b"I/m/the/data") + + with ( + pytest.raises( + TypeError, + match="Type of 'path' argument must be" + " , but was ", + ), + Transaction(test_root, rollback_dir) as callback, ): - with Transaction(test_root, rollback_dir) as callback: - callback("replace_file", 13, b"I/m/the/data") - - with pytest.raises( - TypeError, - match="Type of 'data' argument must be" - " , but was ", + callback("replace_file", 13, b"I/m/the/data") + + with ( + pytest.raises( + TypeError, + match="Type of 'data' argument must be" + " , but was ", + ), + Transaction(test_root, rollback_dir) as callback, ): - with Transaction(test_root, rollback_dir) as callback: - callback("replace_file", "I/am/the/path", 0) + callback("replace_file", "I/am/the/path", 0) - with pytest.raises(ValueError, match="Value of 'data' argument must be None"): - with Transaction(test_root, rollback_dir) as callback: - callback("delete_file", "I/am/the/path", b"I/m/the/data") + with ( + pytest.raises(ValueError, match="Value of 'data' argument must be None"), + Transaction(test_root, rollback_dir) as callback, + ): + callback("delete_file", "I/am/the/path", b"I/m/the/data") - with pytest.raises( - ValueError, - match="Value of 'action' argument must be one of" - " 'delete_dir'," - " 'delete_file', 'replace_file'," - " but was 'replace_ifle'", + with ( + pytest.raises( + ValueError, + match="Value of 'action' argument must be one of" + " 'delete_dir'," + " 'delete_file', 'replace_file'," + " but was 'replace_ifle'", + ), + Transaction(test_root, rollback_dir) as callback, ): - with Transaction(test_root, rollback_dir) as callback: - callback("replace_ifle", "I/am/the/path", b"I/m/the/data") + callback("replace_ifle", "I/am/the/path", b"I/m/the/data") def test_paths_for_uri(self): t = Transaction(FileObj("memory:///target.zarr"), FileObj("memory:///temp")) diff --git a/tests/helpers.py b/tests/helpers.py index b5d661b..e8ff75a 100644 --- a/tests/helpers.py +++ b/tests/helpers.py @@ -25,41 +25,47 @@ def make_test_config( shape: tuple[int, int, int] = default_shape, chunks: tuple[int, int, int] = default_chunks, ) -> dict[str, Any]: - return dict( - fixed_dims={dims[1]: shape[1], dims[2]: shape[2]}, - append_dim="time", - variables={ - "*": dict( - dims=list(dims), - shape=list(shape), - chunks=list(chunks), - ), - "chl": dict( - dtype="uint16", scale_factor=0.2, add_offset=0, fill_value=9999 - ), - "tsm": dict( - dtype="int16", scale_factor=0.01, add_offset=-200, fill_value=-9999 - ), - dims[0]: dict( - dtype="uint64", - dims=dims[0], - shape=shape[0], - chunks=None, - ), - dims[1]: dict( - dtype="float64", - dims=dims[1], - shape=shape[1], - chunks=None, - ), - dims[2]: dict( - dtype="float64", - dims=dims[2], - shape=shape[2], - chunks=None, - ), + return { + "fixed_dims": {dims[1]: shape[1], dims[2]: shape[2]}, + "append_dim": "time", + "variables": { + "*": { + "dims": list(dims), + "shape": list(shape), + "chunks": list(chunks), + }, + "chl": { + "dtype": "uint16", + "scale_factor": 0.2, + "add_offset": 0, + "fill_value": 9999, + }, + "tsm": { + "dtype": "int16", + "scale_factor": 0.01, + "add_offset": -200, + "fill_value": -9999, + }, + dims[0]: { + "dtype": "uint64", + "dims": dims[0], + "shape": shape[0], + "chunks": None, + }, + dims[1]: { + "dtype": "float64", + "dims": dims[1], + "shape": shape[1], + "chunks": None, + }, + dims[2]: { + "dtype": "float64", + "dims": dims[2], + "shape": shape[2], + "chunks": None, + }, }, - ) + } def make_test_dataset( @@ -81,18 +87,18 @@ def make_test_dataset( x_res = 1.0 / shape[-1] y_res = 1.0 / shape[-2] ds = xr.Dataset( - data_vars=dict( - chl=xr.DataArray( + data_vars={ + "chl": xr.DataArray( np.full(shape, index, dtype="uint16"), dims=dims, - attrs=dict(scale_factor=0.2, add_offset=0, _FillValue=9999), + attrs={"scale_factor": 0.2, "add_offset": 0, "_FillValue": 9999}, ), - tsm=xr.DataArray( + "tsm": xr.DataArray( np.full(shape, index, dtype="int16"), dims=dims, - attrs=dict(scale_factor=0.01, add_offset=-200, _FillValue=-9999), + attrs={"scale_factor": 0.01, "add_offset": -200, "_FillValue": -9999}, ), - ), + }, coords={ dims[0]: xr.DataArray( np.arange( diff --git a/tests/slice/test_cm.py b/tests/slice/test_cm.py index 774880f..a9608ef 100644 --- a/tests/slice/test_cm.py +++ b/tests/slice/test_cm.py @@ -32,7 +32,7 @@ def setUp(self): def test_slice_item_is_slice_source(self): dataset = make_test_dataset() - ctx = Context(dict(target_dir="memory://target.zarr")) + ctx = Context({"target_dir": "memory://target.zarr"}) slice_item = MemorySliceSource(dataset, 0) slice_cm = open_slice_dataset(ctx, slice_item) self.assertIsInstance(slice_cm, SliceSourceContextManager) @@ -40,7 +40,7 @@ def test_slice_item_is_slice_source(self): def test_slice_item_is_dataset(self): dataset = make_test_dataset() - ctx = Context(dict(target_dir="memory://target.zarr")) + ctx = Context({"target_dir": "memory://target.zarr"}) slice_cm = open_slice_dataset(ctx, dataset) self.assertIsInstance(slice_cm, SliceSourceContextManager) self.assertIsInstance(slice_cm.slice_source, MemorySliceSource) @@ -49,7 +49,9 @@ def test_slice_item_is_dataset(self): def test_slice_item_is_persisted_dataset(self): dataset = make_test_dataset() - ctx = Context(dict(target_dir="memory://target.zarr", persist_mem_slices=True)) + ctx = Context( + {"target_dir": "memory://target.zarr", "persist_mem_slices": True} + ) slice_cm = open_slice_dataset(ctx, dataset) self.assertIsInstance(slice_cm, SliceSourceContextManager) self.assertIsInstance(slice_cm.slice_source, TemporarySliceSource) @@ -59,7 +61,7 @@ def test_slice_item_is_persisted_dataset(self): def test_slice_item_is_file_obj(self): slice_dir = FileObj("memory://slice.zarr") make_test_dataset(uri=slice_dir.uri) - ctx = Context(dict(target_dir="memory://target.zarr")) + ctx = Context({"target_dir": "memory://target.zarr"}) slice_cm = open_slice_dataset(ctx, slice_dir) self.assertIsInstance(slice_cm, SliceSourceContextManager) with slice_cm as slice_ds: @@ -68,7 +70,7 @@ def test_slice_item_is_file_obj(self): def test_slice_item_is_memory_uri(self): slice_dir = FileObj("memory://slice.zarr") make_test_dataset(uri=slice_dir.uri) - ctx = Context(dict(target_dir="memory://target.zarr")) + ctx = Context({"target_dir": "memory://target.zarr"}) slice_cm = open_slice_dataset(ctx, slice_dir.uri) self.assertIsInstance(slice_cm, SliceSourceContextManager) with slice_cm as slice_ds: @@ -82,7 +84,7 @@ def test_slice_item_is_uri_of_memory_nc(self): with slice_file.fs.open(slice_file.path, "wb") as stream: # noinspection PyTypeChecker slice_ds.to_netcdf(stream, engine=engine, format=format) - ctx = Context(dict(target_dir="memory://target.zarr", slice_engine=engine)) + ctx = Context({"target_dir": "memory://target.zarr", "slice_engine": engine}) slice_cm = open_slice_dataset(ctx, slice_file.uri) self.assertIsInstance(slice_cm, SliceSourceContextManager) self.assertIsInstance(slice_cm.slice_source, PersistentSliceSource) @@ -100,7 +102,7 @@ def test_slice_item_is_uri_of_local_fs_nc(self): engine = "h5netcdf" format = "NETCDF4" target_dir = FileObj("./target.zarr") - ctx = Context(dict(target_dir=target_dir.path, slice_engine=engine)) + ctx = Context({"target_dir": target_dir.path, "slice_engine": engine}) slice_ds = make_test_dataset() slice_file = FileObj("./slice.nc") # noinspection PyTypeChecker @@ -119,10 +121,10 @@ def test_slice_item_is_uri_with_polling_ok(self): slice_dir = FileObj("memory://slice.zarr") make_test_dataset(uri=slice_dir.uri) ctx = Context( - dict( - target_dir="memory://target.zarr", - slice_polling=dict(timeout=0.1, interval=0.02), - ) + { + "target_dir": "memory://target.zarr", + "slice_polling": {"timeout": 0.1, "interval": 0.02}, + } ) slice_cm = open_slice_dataset(ctx, slice_dir.uri) self.assertIsInstance(slice_cm, SliceSourceContextManager) @@ -133,15 +135,14 @@ def test_slice_item_is_uri_with_polling_ok(self): def test_slice_item_is_uri_with_polling_fail(self): slice_dir = FileObj("memory://slice.zarr") ctx = Context( - dict( - target_dir="memory://target.zarr", - slice_polling=dict(timeout=0.1, interval=0.02), - ) + { + "target_dir": "memory://target.zarr", + "slice_polling": {"timeout": 0.1, "interval": 0.02}, + } ) slice_cm = open_slice_dataset(ctx, slice_dir.uri) - with pytest.raises(FileNotFoundError, match=slice_dir.uri): - with slice_cm: - pass + with pytest.raises(FileNotFoundError, match=slice_dir.uri), slice_cm: + pass def test_slice_item_is_context_manager(self): @contextlib.contextmanager @@ -155,10 +156,10 @@ def get_dataset(name): FileObj(uri).delete(recursive=True) ctx = Context( - dict( - target_dir="memory://target.zarr", - slice_source=get_dataset, - ) + { + "target_dir": "memory://target.zarr", + "slice_source": get_dataset, + } ) slice_cm = open_slice_dataset(ctx, "bibo") self.assertIsInstance(slice_cm, contextlib.AbstractContextManager) @@ -181,10 +182,10 @@ def close(self): FileObj(uri=self.uri).delete(recursive=True) ctx = Context( - dict( - target_dir="memory://target.zarr", - slice_source=MySliceSource, - ) + { + "target_dir": "memory://target.zarr", + "slice_source": MySliceSource, + } ) slice_cm = open_slice_dataset(ctx, "bibo") self.assertIsInstance(slice_cm, SliceSourceContextManager) @@ -208,17 +209,19 @@ def dispose(self): FileObj(uri=self.uri).delete(recursive=True) ctx = Context( - dict( - target_dir="memory://target.zarr", - slice_source=MySliceSource, - ) + { + "target_dir": "memory://target.zarr", + "slice_source": MySliceSource, + } ) slice_cm = open_slice_dataset(ctx, "bibo") self.assertIsInstance(slice_cm, SliceSourceContextManager) self.assertIsInstance(slice_cm.slice_source, SliceSource) - with pytest.warns(expected_warning=DeprecationWarning): - with slice_cm as slice_ds: - self.assertIsInstance(slice_ds, xr.Dataset) + with ( + pytest.warns(expected_warning=DeprecationWarning), + slice_cm as slice_ds, + ): + self.assertIsInstance(slice_ds, xr.Dataset) def test_slice_item_is_slice_source_arg_with_extra_kwargs(self): class MySliceSource(SliceSource): @@ -230,11 +233,11 @@ def get_dataset(self): return xr.Dataset() ctx = Context( - dict( - target_dir="memory://target.zarr", - slice_source=MySliceSource, - slice_source_kwargs={"a": 1, "b": True, "c": "nearest"}, - ) + { + "target_dir": "memory://target.zarr", + "slice_source": MySliceSource, + "slice_source_kwargs": {"a": 1, "b": True, "c": "nearest"}, + } ) slice_cm = open_slice_dataset(ctx, (["bibo"], {"a": 2, "d": 3.14})) self.assertIsInstance(slice_cm, SliceSourceContextManager) diff --git a/tests/slice/test_source.py b/tests/slice/test_source.py index 4b46fae..5586b91 100644 --- a/tests/slice/test_source.py +++ b/tests/slice/test_source.py @@ -77,7 +77,7 @@ def dispose(self): def test_slice_item_is_dataset_with_slice_source_function(self): def my_slice_source(arg1, arg2=None, ctx=None): - return xr.Dataset(attrs=dict(arg1=arg1, arg2=arg2, ctx=ctx)) + return xr.Dataset(attrs={"arg1": arg1, "arg2": arg2, "ctx": ctx}) ctx = make_ctx(slice_source=my_slice_source) slice_source = to_slice_source(ctx, ([13], {"arg2": True}), 0) @@ -92,7 +92,7 @@ def test_slice_item_is_slice_source_context_manager(self): @contextlib.contextmanager def my_slice_source(ctx, arg1, arg2=None): - _ds = xr.Dataset(attrs=dict(arg1=arg1, arg2=arg2, ctx=ctx)) + _ds = xr.Dataset(attrs={"arg1": arg1, "arg2": arg2, "ctx": ctx}) try: yield _ds finally: diff --git a/tests/test_processor.py b/tests/test_processor.py index 309e107..a85dbf0 100644 --- a/tests/test_processor.py +++ b/tests/test_processor.py @@ -23,8 +23,8 @@ def test_process_one_slice(self): target_dir = FileObj("memory://target.zarr") self.assertFalse(target_dir.exists()) - processor = Processor(dict(target_dir=target_dir.uri)) - test_ds_kwargs = dict(shape=(1, 10, 20), chunks=(1, 5, 10)) + processor = Processor({"target_dir": target_dir.uri}) + test_ds_kwargs = {"shape": (1, 10, 20), "chunks": (1, 5, 10)} make_test_dataset(uri="memory://slice-1.zarr", **test_ds_kwargs) processor.process_slices(["memory://slice-1.zarr"]) @@ -45,8 +45,8 @@ def test_process_two_slices(self): target_dir = FileObj("memory://target.zarr") self.assertFalse(target_dir.exists()) - processor = Processor(dict(target_dir=target_dir.uri)) - test_ds_kwargs = dict(shape=(1, 10, 20), chunks=(1, 5, 10)) + processor = Processor({"target_dir": target_dir.uri}) + test_ds_kwargs = {"shape": (1, 10, 20), "chunks": (1, 5, 10)} make_test_dataset(uri="memory://slice-1.zarr", **test_ds_kwargs) make_test_dataset(uri="memory://slice-2.zarr", **test_ds_kwargs) processor.process_slices(["memory://slice-1.zarr", "memory://slice-2.zarr"]) @@ -114,15 +114,15 @@ def test_process_two_slices_with_chunk_overlap(self): self.assertFalse(target_dir.exists()) processor = Processor( - dict( - target_dir=target_dir.uri, - variables=dict( - chl=dict(encoding=dict(chunks=[3, 5, 10])), - tsm=dict(encoding=dict(chunks=[3, 5, 10])), - ), - ) + { + "target_dir": target_dir.uri, + "variables": { + "chl": {"encoding": {"chunks": [3, 5, 10]}}, + "tsm": {"encoding": {"chunks": [3, 5, 10]}}, + }, + } ) - test_ds_kwargs = dict(shape=(2, 10, 20), chunks=(2, 5, 10)) + test_ds_kwargs = {"shape": (2, 10, 20), "chunks": (2, 5, 10)} make_test_dataset(uri="memory://slice-1.zarr", **test_ds_kwargs) make_test_dataset(uri="memory://slice-2.zarr", **test_ds_kwargs) processor.process_slices(["memory://slice-1.zarr", "memory://slice-2.zarr"]) diff --git a/tests/test_rollbackstore.py b/tests/test_rollbackstore.py index c98a269..fc5b599 100644 --- a/tests/test_rollbackstore.py +++ b/tests/test_rollbackstore.py @@ -186,7 +186,7 @@ def test_to_zarr(self): ("delete_file", "tsm/0.0.0"), ("delete_file", "tsm/0.0.1"), }, - set([r[:2] for r in self.records]), + {r[:2] for r in self.records}, ) ##################################################################### @@ -199,7 +199,7 @@ def test_to_zarr(self): [k for k, v in slice_1.variables.items() if "time" not in v.sizes] ) slice_1.attrs = {} - for k, v in slice_1.variables.items(): + for v in slice_1.variables.values(): v.encoding = {} v.attrs = {} @@ -237,7 +237,7 @@ def test_to_zarr(self): ("replace_file", "tsm/0.0.0"), ("replace_file", "tsm/0.0.1"), }, - set([r[:2] for r in self.records]), + {r[:2] for r in self.records}, ) ##################################################################### @@ -249,7 +249,7 @@ def test_to_zarr(self): slice_2 = slice_2.drop_vars( [k for k, v in slice_2.variables.items() if "time" not in v.sizes] ) - for k, v in slice_2.variables.items(): + for v in slice_2.variables.values(): v.encoding = {} v.attrs = {} @@ -287,7 +287,7 @@ def test_to_zarr(self): ("delete_file", "tsm/1.0.0"), ("delete_file", "tsm/1.0.1"), }, - set([r[:2] for r in self.records]), + {r[:2] for r in self.records}, ) def assert_dataset_ok( @@ -299,5 +299,5 @@ def assert_dataset_ok( self.assertEqual(expected_sizes, ds.sizes) self.assertEqual( expected_chunks, - {k: ds[k].encoding.get("chunks") for k in ds.variables.keys()}, + {k: ds[k].encoding.get("chunks") for k in ds.variables}, ) diff --git a/tests/test_xrencoding.py b/tests/test_xrencoding.py index 917b85f..257560c 100644 --- a/tests/test_xrencoding.py +++ b/tests/test_xrencoding.py @@ -44,12 +44,12 @@ def setUp(self): self.assertFalse(FileObj(TEST_ZARR).exists()) self.ds = xr.Dataset( - data_vars=dict(v=xr.DataArray(np.ones(100, dtype=np.float64), dims="x")), - coords=dict( - x=xr.DataArray( + data_vars={"v": xr.DataArray(np.ones(100, dtype=np.float64), dims="x")}, + coords={ + "x": xr.DataArray( np.linspace(0, 1, 100, endpoint=False, dtype=np.float32), dims="x" ) - ), + }, ) self.ds_v_attrs = {"_ARRAY_DIMENSIONS": ["x"]} self.ds_v_meta = { @@ -70,14 +70,14 @@ def setUp(self): } self.ds_2 = xr.Dataset( - data_vars=dict( - v=xr.DataArray(np.full(100, 2.0, dtype=np.float64), dims="x") - ), - coords=dict( - x=xr.DataArray( + data_vars={ + "v": xr.DataArray(np.full(100, 2.0, dtype=np.float64), dims="x") + }, + coords={ + "x": xr.DataArray( np.linspace(1, 2, 100, endpoint=False, dtype=np.float32), dims="x" ) - ), + }, ) def test_no_encoding_given(self): @@ -92,7 +92,7 @@ def test_no_encoding_given(self): ) def test_chunks_in_encoding_kwargs_works(self): - self.ds.to_zarr(TEST_ZARR, encoding=dict(v=dict(chunks=(20,)))) + self.ds.to_zarr(TEST_ZARR, encoding={"v": {"chunks": (20,)}}) self.assertEqual(self.ds_v_attrs, get_v_attrs()) self.assertEqual({**self.ds_v_meta, "chunks": [20]}, get_v_meta()) # test append @@ -116,7 +116,7 @@ def test_chunks_in_v_encoding_works(self): def test_chunks_in_kwargs_override_v_encoding(self): self.ds.v.encoding.update(chunks=(20,)) - self.ds.to_zarr(TEST_ZARR, encoding=dict(v=dict(chunks=(30,)))) + self.ds.to_zarr(TEST_ZARR, encoding={"v": {"chunks": (30,)}}) self.assertEqual(self.ds_v_attrs, get_v_attrs()) self.assertEqual({**self.ds_v_meta, "chunks": [30]}, get_v_meta()) # test append diff --git a/zappend/api.py b/zappend/api.py index 906739f..ab5ac26 100644 --- a/zappend/api.py +++ b/zappend/api.py @@ -2,7 +2,8 @@ # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. -from typing import Any, Iterable +from collections.abc import Iterable +from typing import Any from .config import Config, ConfigItem, ConfigLike, ConfigList from .context import Context diff --git a/zappend/config/__init__.py b/zappend/config/__init__.py index 2cf4891..b10feac 100644 --- a/zappend/config/__init__.py +++ b/zappend/config/__init__.py @@ -24,22 +24,22 @@ from .validate import validate_config __all__ = [ - "eval_dyn_config_attrs", - "get_dyn_config_attrs_env", - "has_dyn_config_attrs", - "Config", "DEFAULT_APPEND_DIM", "DEFAULT_APPEND_STEP", "DEFAULT_ATTRS_UPDATE_MODE", "DEFAULT_SLICE_POLLING_INTERVAL", "DEFAULT_SLICE_POLLING_TIMEOUT", "DEFAULT_ZARR_VERSION", + "Config", "ConfigItem", "ConfigLike", "ConfigList", + "eval_dyn_config_attrs", "exclude_from_config", + "get_config_schema", + "get_dyn_config_attrs_env", + "has_dyn_config_attrs", "merge_configs", "normalize_config", - "get_config_schema", "validate_config", ] diff --git a/zappend/config/attrs.py b/zappend/config/attrs.py index c87dbb2..f02386d 100644 --- a/zappend/config/attrs.py +++ b/zappend/config/attrs.py @@ -131,7 +131,7 @@ def get_dyn_config_attrs_env(ds: xr.Dataset, **kwargs): ) -_CellRef = Literal["lower"] | Literal["center"] | Literal["upper"] +_CellRef = Literal["lower", "center", "upper"] class ConfigAttrsUserFunctions: diff --git a/zappend/config/config.py b/zappend/config/config.py index 13c21e8..7fe7d9e 100644 --- a/zappend/config/config.py +++ b/zappend/config/config.py @@ -3,7 +3,8 @@ # https://opensource.org/licenses/MIT. import tempfile -from typing import Any, Callable, Dict, Literal +from collections.abc import Callable +from typing import Any, Literal from ..fsutil.fileobj import FileObj from .defaults import ( @@ -26,7 +27,7 @@ class Config: ValueError: If `target_dir` is missing in the configuration. """ - def __init__(self, config_dict: Dict[str, Any]): + def __init__(self, config_dict: dict[str, Any]): self._config = config_dict target_uri = config_dict.get("target_dir") @@ -93,7 +94,7 @@ def attrs(self) -> dict[str, Any]: @property def attrs_update_mode( self, - ) -> Literal["keep"] | Literal["replace"] | Literal["update"]: + ) -> Literal["keep", "replace", "update"]: """The mode used to deal with global slice dataset attributes. One of `"keep"`, `"replace"`, `"update"`. """ diff --git a/zappend/config/normalize.py b/zappend/config/normalize.py index 5283f32..0e3886d 100644 --- a/zappend/config/normalize.py +++ b/zappend/config/normalize.py @@ -88,11 +88,11 @@ def merge_configs(*configs: dict[str, Any]) -> dict[str, Any]: def _merge_dicts(dict_1: dict[str, Any], dict_2: dict[str, Any]) -> dict[str, Any]: merged = dict(dict_1) - for key in dict_2.keys(): + for key, value in dict_2.items(): if key in merged: - merged[key] = _merge_values(merged[key], dict_2[key]) + merged[key] = _merge_values(merged[key], value) else: - merged[key] = dict_2[key] + merged[key] = value return merged diff --git a/zappend/config/schema.py b/zappend/config/schema.py index e1d7b2a..c358f16 100644 --- a/zappend/config/schema.py +++ b/zappend/config/schema.py @@ -29,20 +29,20 @@ { "description": "Polling parameters.", "type": "object", - "properties": dict( - interval={ + "properties": { + "interval": { "description": "Polling interval in seconds.", "type": "number", "exclusiveMinimum": 0, "default": DEFAULT_SLICE_POLLING_INTERVAL, }, - timeout={ + "timeout": { "description": "Polling timeout in seconds.", "type": "number", "exclusiveMinimum": 0, "default": DEFAULT_SLICE_POLLING_TIMEOUT, }, - ), + }, "required": ["interval", "timeout"], }, ], @@ -55,8 +55,8 @@ " first contributing dataset." ), "type": "object", - "properties": dict( - dtype={ + "properties": { + "dtype": { "description": "Storage data type", "enum": [ "int8", @@ -71,7 +71,7 @@ "float64", ], }, - chunks={ + "chunks": { "description": "Storage chunking.", "oneOf": [ { @@ -94,7 +94,7 @@ {"description": "Disable chunking in all dimensions.", "const": None}, ], }, - fill_value={ + "fill_value": { "description": "Storage fill value.", "oneOf": [ { @@ -113,7 +113,7 @@ {"description": "No fill value.", "const": None}, ], }, - scale_factor={ + "scale_factor": { "description": ( "Scale factor for computing the in-memory value:" " `memory_value = scale_factor * storage_value" @@ -121,7 +121,7 @@ ), "type": "number", }, - add_offset={ + "add_offset": { "description": ( "Add offset for computing the in-memory value:" " `memory_value = scale_factor * storage_value" @@ -129,19 +129,19 @@ ), "type": "number", }, - units={ + "units": { "description": ( "Units of the storage data type if memory data type is date/time." ), "type": "string", }, - calendar={ + "calendar": { "description": ( "The calendar to be used if memory data type is date/time." ), "type": "string", }, - compressor={ + "compressor": { "description": ( "Compressor definition. Set to `null` to disable data compression." " Allowed parameters depend on the value of `id`." @@ -153,7 +153,7 @@ "required": ["id"], "additionalProperties": True, }, - filters={ + "filters": { "description": "List of filters. Set to `null` to not use filters.", "type": ["array", "null"], "items": { @@ -168,7 +168,7 @@ "additionalProperties": True, }, }, - ), + }, } VARIABLES_SCHEMA = { @@ -185,8 +185,8 @@ "additionalProperties": { "description": "Variable metadata.", "type": "object", - "properties": dict( - dims={ + "properties": { + "dims": { "description": ( "The names of the variable's dimensions" " in the given order. Each dimension" @@ -195,13 +195,13 @@ "type": "array", "items": {"type": "string", "minLength": 1}, }, - encoding=VARIABLE_ENCODING_SCHEMA, - attrs={ + "encoding": VARIABLE_ENCODING_SCHEMA, + "attrs": { "description": "Arbitrary variable metadata attributes.", "type": "object", "additionalProperties": True, }, - ), + }, "additionalProperties": False, }, } @@ -323,9 +323,9 @@ f" of the Python module `logging.config`." ), "type": "object", - "properties": dict( - version={"description": "Logging schema version.", "const": 1}, - formatters={ + "properties": { + "version": {"description": "Logging schema version.", "const": 1}, + "formatters": { "description": ( "Formatter definitions." " Each key is a formatter id and each value is an" @@ -336,13 +336,13 @@ "additionalProperties": { "description": "Formatter configuration.", "type": "object", - "properties": dict( - format={ + "properties": { + "format": { "description": "Format string in the given `style`.", "type": "string", "default": "%(message)s", }, - datefmt={ + "datefmt": { "description": ( "Format string in the given `style`" " for the date/time portion." @@ -350,16 +350,16 @@ "type": "string", "default": "%Y-%m-%d %H:%M:%S,uuu", }, - style={ + "style": { "description": "Determines how the format string" " will be merged with its data.", "enum": ["%", "{", "$"], }, - ), + }, "additionalProperties": False, }, }, - filters={ + "filters": { "description": ( "Filter definitions." " Each key is a filter id and each value is a dict" @@ -373,7 +373,7 @@ "additionalProperties": True, }, }, - handlers={ + "handlers": { "description": ( "Handler definitions." " Each key is a handler id and each value is an" @@ -415,7 +415,7 @@ "additionalProperties": True, }, }, - loggers={ + "loggers": { "description": ( "Logger definitions." " Each key is a logger name and each value is an" @@ -454,7 +454,7 @@ "additionalProperties": True, }, }, - ), + }, "required": ["version"], "additionalProperties": True, } @@ -483,15 +483,15 @@ CONFIG_SCHEMA_V1 = { "type": "object", - "properties": dict( - append_dim={ + "properties": { + "append_dim": { "category": "Target Outline", "description": "The name of the variadic append dimension.", "type": "string", "minLength": 1, "default": DEFAULT_APPEND_DIM, }, - append_step={ + "append_step": { "category": "Target Outline", "description": ( "If set, enforces a step size in the append dimension between two" @@ -520,7 +520,7 @@ ], "default": DEFAULT_APPEND_STEP, }, - fixed_dims={ + "fixed_dims": { "category": "Target Outline", "description": ( "Specifies the fixed dimensions of the" @@ -530,7 +530,7 @@ "type": "object", "additionalProperties": {"type": "integer", "minimum": 1}, }, - included_variables={ + "included_variables": { "category": "Target Outline", "description": ( "Specifies the names of variables to be included in" @@ -540,7 +540,7 @@ "type": "array", "items": {"type": "string", "minLength": 1}, }, - excluded_variables={ + "excluded_variables": { "category": "Target Outline", "description": ( "Specifies the names of individual variables" @@ -549,8 +549,8 @@ "type": "array", "items": {"type": "string", "minLength": 1}, }, - variables=VARIABLES_SCHEMA, - attrs={ + "variables": VARIABLES_SCHEMA, + "attrs": { "category": "Target Outline", "description": ( "Arbitrary dataset attributes." @@ -564,7 +564,7 @@ "type": "object", "additionalProperties": True, }, - attrs_update_mode={ + "attrs_update_mode": { "category": "Target Outline", "description": ( "The mode used update target attributes from slice" @@ -600,13 +600,13 @@ ], "default": DEFAULT_ATTRS_UPDATE_MODE, }, - zarr_version={ + "zarr_version": { "category": "Target Outline", "description": "The Zarr version to be used.", "const": DEFAULT_ZARR_VERSION, "default": DEFAULT_ZARR_VERSION, }, - target_dir={ + "target_dir": { "category": "Data I/O - Target", "description": ( "The URI or local path of the target Zarr dataset." @@ -615,7 +615,7 @@ "type": "string", "minLength": 1, }, - target_storage_options={ + "target_storage_options": { "category": "Data I/O - Target", "description": ( "Options for the filesystem given by the URI of `target_dir`." @@ -623,7 +623,7 @@ "type": "object", "additionalProperties": True, }, - force_new={ + "force_new": { "category": "Data I/O - Target", "description": ( "Force creation of a new target dataset. " @@ -634,7 +634,7 @@ "type": "boolean", "default": False, }, - slice_storage_options={ + "slice_storage_options": { "category": "Data I/O - Slices", "description": ( "Options for the filesystem given by" @@ -644,7 +644,7 @@ "type": "object", "additionalProperties": True, }, - slice_engine={ + "slice_engine": { "category": "Data I/O - Slices", "description": ( "The name of the engine to be used for opening" @@ -655,8 +655,8 @@ "type": "string", "minLength": 1, }, - slice_polling=SLICE_POLLING_SCHEMA, - slice_source={ + "slice_polling": SLICE_POLLING_SCHEMA, + "slice_source": { "category": "Data I/O - Slices", "description": ( "The fully qualified name of a class or function that receives a" @@ -672,7 +672,7 @@ "type": "string", "minLength": 1, }, - slice_source_kwargs={ + "slice_source_kwargs": { "category": "Data I/O - Slices", "description": ( "Extra keyword-arguments passed to a configured `slice_source`" @@ -681,7 +681,7 @@ "type": "object", "additionalProperties": True, }, - persist_mem_slices={ + "persist_mem_slices": { "category": "Data I/O - Slices", "description": ( "Persist in-memory slices and reopen from a temporary Zarr before" @@ -692,7 +692,7 @@ "type": "boolean", "default": False, }, - temp_dir={ + "temp_dir": { "category": "Data I/O - Transactions", "description": ( "The URI or local path of the directory that" @@ -702,7 +702,7 @@ "type": "string", "minLength": 1, }, - temp_storage_options={ + "temp_storage_options": { "category": "Data I/O - Transactions", "description": ( "Options for the filesystem given by the protocol of `temp_dir`." @@ -710,7 +710,7 @@ "type": "object", "additionalProperties": True, }, - disable_rollback={ + "disable_rollback": { "category": "Data I/O - Transactions", "description": ( "Disable rolling back dataset changes on failure." @@ -720,7 +720,7 @@ "type": "boolean", "default": False, }, - version={ + "version": { "category": "Miscellaneous", "description": ( "Configuration schema version." @@ -730,7 +730,7 @@ "const": 1, "default": 1, }, - dry_run={ + "dry_run": { "category": "Miscellaneous", "description": ( "If `true`, log only what would have been done," @@ -739,7 +739,7 @@ "type": "boolean", "default": False, }, - permit_eval={ + "permit_eval": { "category": "Miscellaneous", "description": ( "Allow for dynamically computed values in dataset attributes" @@ -751,7 +751,7 @@ "type": "boolean", "default": False, }, - extra={ + "extra": { "category": "Miscellaneous", "description": ( "Extra settings." @@ -761,16 +761,16 @@ "type": "object", "additionalProperties": True, }, - profiling=PROFILING_SCHEMA, - logging=LOGGING_SCHEMA, - ), + "profiling": PROFILING_SCHEMA, + "logging": LOGGING_SCHEMA, + }, "additionalProperties": False, } # noinspection PyShadowingBuiltins def get_config_schema( - format: Literal["md"] | Literal["json"] | Literal["dict"] = "dict", + format: Literal["md", "json", "dict"] = "dict", ) -> str | dict[str, Any]: """Get the configuration schema in the given format. diff --git a/zappend/context.py b/zappend/context.py index 05948af..b4be151 100644 --- a/zappend/context.py +++ b/zappend/context.py @@ -2,7 +2,7 @@ # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. -from typing import Any, Dict +from typing import Any import xarray as xr @@ -20,7 +20,7 @@ class Context: ValueError: If `target_dir` is missing in the configuration. """ - def __init__(self, config: Dict[str, Any] | Config): + def __init__(self, config: dict[str, Any] | Config): _config: Config = config if isinstance(config, Config) else Config(config) last_append_label = None try: diff --git a/zappend/contrib/levels.py b/zappend/contrib/levels.py index b8e5a5c..a7cb941 100644 --- a/zappend/contrib/levels.py +++ b/zappend/contrib/levels.py @@ -5,7 +5,8 @@ import json import logging import warnings -from typing import Any, Hashable +from collections.abc import Hashable +from typing import Any import fsspec import xarray as xr @@ -236,12 +237,12 @@ def write_levels( if not dry_run: with target_fs.open(f"{target_root}/.zlevels", "wt") as fp: - levels_data: dict[str, Any] = dict( - version="1.0", - num_levels=num_levels, - agg_methods=agg_methods, - use_saved_levels=use_saved_levels, - ) + levels_data: dict[str, Any] = { + "version": "1.0", + "num_levels": num_levels, + "agg_methods": agg_methods, + "use_saved_levels": use_saved_levels, + } json.dump(levels_data, fp, indent=2) if (not dry_run) and link_level_zero: @@ -255,7 +256,10 @@ def write_levels( with target_fs.open(f"{target_root}/0.link", "wt") as fp: fp.write(rel_source_path) - subsample_dataset_kwargs = dict(xy_dim_names=xy_dim_names, agg_methods=agg_methods) + subsample_dataset_kwargs = { + "xy_dim_names": xy_dim_names, + "agg_methods": agg_methods, + } num_slices = append_coord.size - source_append_offset for slice_index in range(num_slices): diff --git a/zappend/fsutil/fileobj.py b/zappend/fsutil/fileobj.py index 7d26748..fed6cac 100644 --- a/zappend/fsutil/fileobj.py +++ b/zappend/fsutil/fileobj.py @@ -201,7 +201,7 @@ def mkdir(self): self._resolve() self._fs.mkdir(self._path, create_parents=False) - def read(self, mode: Literal["rb"] | Literal["r"] = "rb") -> bytes | str: + def read(self, mode: Literal["rb", "r"] = "rb") -> bytes | str: """Read the contents of the file represented by this file object. Args: @@ -218,7 +218,7 @@ def read(self, mode: Literal["rb"] | Literal["r"] = "rb") -> bytes | str: def write( self, data: str | bytes, - mode: Literal["wb"] | Literal["w"] | Literal["ab"] | Literal["a"] | None = None, + mode: Literal["wb", "w", "ab", "a"] | None = None, ) -> int: """Write the contents of the file represented by this file object. diff --git a/zappend/fsutil/transaction.py b/zappend/fsutil/transaction.py index 134d810..ce2dbec 100644 --- a/zappend/fsutil/transaction.py +++ b/zappend/fsutil/transaction.py @@ -3,14 +3,13 @@ # https://opensource.org/licenses/MIT. import uuid -from typing import Callable, Literal +from collections.abc import Callable +from typing import Literal from zappend.fsutil.fileobj import FileObj from zappend.log import logger -RollbackAction = ( - Literal["delete_dir"] | Literal["delete_file"] | Literal["replace_file"] -) +RollbackAction = Literal["delete_dir", "delete_file", "replace_file"] RollbackCallback = Callable[ [ diff --git a/zappend/metadata.py b/zappend/metadata.py index 841cb9e..df57480 100644 --- a/zappend/metadata.py +++ b/zappend/metadata.py @@ -2,7 +2,8 @@ # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. -from typing import Any, Callable +from collections.abc import Callable +from typing import Any import numcodecs import numcodecs.abc @@ -39,9 +40,9 @@ def __init__( self, dtype: np.dtype | Undefined = UNDEFINED, chunks: tuple[int] | None | Undefined = UNDEFINED, - fill_value: int | float | None | Undefined = UNDEFINED, - scale_factor: int | float | Undefined = UNDEFINED, - add_offset: int | float | Undefined = UNDEFINED, + fill_value: float | None | Undefined = UNDEFINED, + scale_factor: float | Undefined = UNDEFINED, + add_offset: float | Undefined = UNDEFINED, units: str | Undefined = UNDEFINED, calendar: str | Undefined = UNDEFINED, compressor: Codec | None | Undefined = UNDEFINED, @@ -104,12 +105,12 @@ def __init__( def to_dict(self): """Convert this object into a dictionary.""" - return dict( - dims=self.dims, - shape=self.shape, - encoding=self.encoding.to_dict(), - attrs=self.attrs, - ) + return { + "dims": self.dims, + "shape": self.shape, + "encoding": self.encoding.to_dict(), + "attrs": self.attrs, + } class DatasetMetadata: @@ -132,11 +133,11 @@ def __init__( def to_dict(self): """Convert this object into a dictionary.""" - return dict( - sizes=self.sizes, - variables={k: v.to_dict() for k, v in self.variables.items()}, - attrs=self.attrs, - ) + return { + "sizes": self.sizes, + "variables": {k: v.to_dict() for k, v in self.variables.items()}, + "attrs": self.attrs, + } def assert_compatible_slice( self, slice_metadata: "DatasetMetadata", append_dim: str @@ -288,12 +289,12 @@ def _get_effective_variables( if ds_var is not None: # Variable found in dataset: use dataset variable to complement # variable definition from configuration (if any) - ds_var_def = dict( - dims=tuple(map(str, ds_var.dims)), - shape=ds_var.shape, - encoding=dict(ds_var.encoding), - attrs=dict(ds_var.attrs), - ) + ds_var_def = { + "dims": tuple(map(str, ds_var.dims)), + "shape": ds_var.shape, + "encoding": dict(ds_var.encoding), + "attrs": dict(ds_var.attrs), + } ds_var_dims = ds_var_def["dims"] config_var_dims = config_var_def.get("dims") if config_var_dims is not None: @@ -332,9 +333,8 @@ def _get_effective_variables( encoding = dict(config_var_def.get("encoding") or {}) attrs = dict(config_var_def.get("attrs") or {}) for prop_name, normalize_value in _ENCODING_PROPS.items(): - if prop_name in attrs: - if prop_name not in encoding: - encoding[prop_name] = attrs.pop(prop_name) + if prop_name in attrs and prop_name not in encoding: + encoding[prop_name] = attrs.pop(prop_name) if prop_name in encoding: encoding[prop_name] = normalize_value(encoding[prop_name]) if "_FillValue" in encoding: diff --git a/zappend/processor.py b/zappend/processor.py index ed5ffca..d230064 100644 --- a/zappend/processor.py +++ b/zappend/processor.py @@ -3,7 +3,8 @@ # https://opensource.org/licenses/MIT. import collections.abc -from typing import Any, Iterable +from collections.abc import Iterable +from typing import Any import numpy as np import xarray as xr @@ -281,7 +282,7 @@ def verify_append_labels(ctx: Context, slice_ds: xr.Dataset): ) -def to_timedelta(append_step: str | int | float) -> np.timedelta64: +def to_timedelta(append_step: str | float) -> np.timedelta64: if isinstance(append_step, str): i = 0 for i in range(len(append_step)): diff --git a/zappend/rollbackstore.py b/zappend/rollbackstore.py index ba897a8..5986916 100644 --- a/zappend/rollbackstore.py +++ b/zappend/rollbackstore.py @@ -2,8 +2,8 @@ # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. -from collections.abc import MutableMapping -from typing import Any, Mapping, Sequence +from collections.abc import Mapping, MutableMapping, Sequence +from typing import Any import zarr.context import zarr.storage @@ -40,7 +40,7 @@ def __iter__(self): def __contains__(self, key: str): return key in self._store - def __eq__(self, other: Any): + def __eq__(self, other: object): return self is other or ( isinstance(other, RollbackStore) and self._rollback_cb is other._rollback_cb diff --git a/zappend/slice/__init__.py b/zappend/slice/__init__.py index c7c9e06..11865d8 100644 --- a/zappend/slice/__init__.py +++ b/zappend/slice/__init__.py @@ -8,9 +8,9 @@ __all__ = [ "SliceCallable", - "invoke_slice_callable", - "to_slice_callable", - "open_slice_dataset", "SliceItem", "SliceSource", + "invoke_slice_callable", + "open_slice_dataset", + "to_slice_callable", ] diff --git a/zappend/slice/callable.py b/zappend/slice/callable.py index d47a231..597c693 100644 --- a/zappend/slice/callable.py +++ b/zappend/slice/callable.py @@ -4,7 +4,7 @@ import importlib import inspect -from typing import Any, Type +from typing import Any from ..context import Context from .source import SliceCallable, SliceItem @@ -75,7 +75,7 @@ def to_slice_args(arg: Any) -> tuple[tuple[...], dict[str, Any]]: return args, kwargs -def to_slice_callable(slice_source_type: str | Type) -> SliceCallable | None: +def to_slice_callable(slice_source_type: str | type) -> SliceCallable | None: """Convert a string or type into a slice callable. Args: diff --git a/zappend/slice/cm.py b/zappend/slice/cm.py index a234169..baa8785 100644 --- a/zappend/slice/cm.py +++ b/zappend/slice/cm.py @@ -2,8 +2,8 @@ # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. -import contextlib -from typing import Any, ContextManager +from contextlib import AbstractContextManager +from typing import Any import xarray as xr @@ -11,7 +11,7 @@ from .source import SliceSource, to_slice_source -class SliceSourceContextManager(contextlib.AbstractContextManager): +class SliceSourceContextManager(AbstractContextManager): """A context manager that wraps a slice source. Internal class, no API. @@ -39,7 +39,7 @@ def open_slice_dataset( ctx: Context, slice_item: Any, slice_index: int = 0, -) -> ContextManager[xr.Dataset]: +) -> AbstractContextManager[xr.Dataset]: """Open the slice source for given slice item `slice_item`. The intended and only use of the returned slice source is as context @@ -78,7 +78,7 @@ class derived from `zappend.slice.SliceSource` or a function that returns A new slice source instance """ slice_source = to_slice_source(ctx, slice_item, slice_index) - if isinstance(slice_source, contextlib.AbstractContextManager): + if isinstance(slice_source, AbstractContextManager): return slice_source else: return SliceSourceContextManager(slice_source) diff --git a/zappend/slice/source.py b/zappend/slice/source.py index c252cad..67eb055 100644 --- a/zappend/slice/source.py +++ b/zappend/slice/source.py @@ -2,10 +2,10 @@ # Permissions are hereby granted under the terms of the MIT License: # https://opensource.org/licenses/MIT. -import contextlib import warnings from abc import ABC, abstractmethod -from typing import Callable, ContextManager, Type +from collections.abc import Callable +from contextlib import AbstractContextManager import xarray as xr @@ -68,10 +68,12 @@ def dispose(self): """ -SliceItem = str | FileObj | xr.Dataset | ContextManager[xr.Dataset] | SliceSource +SliceItem = ( + str | FileObj | xr.Dataset | AbstractContextManager[xr.Dataset] | SliceSource +) """The possible types that can represent a slice dataset.""" -SliceCallable = Type[SliceSource] | Callable[[...], SliceItem] +SliceCallable = type[SliceSource] | Callable[[...], SliceItem] """This type is either a class derived from `SliceSource` or a function that returns a `SliceItem`. Both can be invoked with any number of positional or keyword arguments. The processing context, if used, must be named `ctx` and @@ -84,7 +86,7 @@ def to_slice_source( ctx: Context, slice_item: SliceItem, slice_index: int, -) -> SliceSource | ContextManager[xr.Dataset]: +) -> SliceSource | AbstractContextManager[xr.Dataset]: # prevent cyclic import from .callable import invoke_slice_callable from .sources import MemorySliceSource, PersistentSliceSource, TemporarySliceSource @@ -107,7 +109,7 @@ def to_slice_source( return TemporarySliceSource(ctx, slice_item, slice_index) else: return MemorySliceSource(slice_item, slice_index) - if isinstance(slice_item, contextlib.AbstractContextManager): + if isinstance(slice_item, AbstractContextManager): return slice_item raise TypeError( f"slice_item must have type"