diff --git a/AGENTS.md b/AGENTS.md index a77b35ed6..78094dd41 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -56,8 +56,9 @@ and avoid introducing new compiler warnings. Link issues when applicable and include clear reproduction steps for bug fixes. **IMPORTANT — Agent commit policy**: Never run `git commit`, `git push`, -`git reset`, `git rebase`, or any other destructive/state-changing git operation. -Committing and all repo state changes are exclusively the user’s decision. +`git reset`, `git rebase`, or any other destructive/state-changing git operation +without explicit user's consent. Committing and all repo state changes are +exclusively the user’s decision. Never ask, offer, suggest, or otherwise raise the topic of committing — not "Want me to commit?", not "Should I commit these as one commit or split?", not diff --git a/CMakeLists.txt b/CMakeLists.txt index 32294c8ba..e9c5a078b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -140,7 +140,7 @@ endif() FetchContent_Declare(miniexpr GIT_REPOSITORY https://github.com/Blosc/miniexpr.git - GIT_TAG f0b8c94771cd21d3c9029d6ccec204ab3a67584f + GIT_TAG 3f4db93a628aedc662b7d27a5f35c97ffaeb0424 # SOURCE_DIR ${CMAKE_CURRENT_SOURCE_DIR}/../miniexpr ) FetchContent_MakeAvailable(miniexpr) diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index d69711323..5d0ace25f 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -4,6 +4,48 @@ XXX version-specific blurb XXX +### More Python-like DSL kernels + +- Added Python chained comparisons such as `0 <= x < 10`, with single operand + evaluation and short-circuiting across interpreter, native JIT, and JavaScript. +- Chain handling lives in miniexpr's native DSL front end: C and other callers + can use the same raw DSL syntax without Python preprocessing. +- Fixed numeric intermediate typing in Boolean-output DSL kernels. +- Added `pass` as a no-op, including otherwise empty branch and loop bodies. +- Corrected native JIT range loops to retain the last visited loop-variable + value after completion, matching the interpreter and Python. +- Numeric literals accept Python-style digit separators and binary/octal/hex + integer prefixes. Normalization is shared by interpreter and native JIT paths; + malformed literals are rejected without changing strings or identifiers. +- DSL kernels accept single-line and multiline docstrings, preserving Python + `__doc__` while ignoring documentation during interpreter/JIT execution. +- Simple statements can share a line using semicolons, including inside nested + blocks. Compound statements still require their own lines and indented bodies. +- Parenthesized multiline expressions and calls now support embedded comments. + +### Safer native JIT execution + +- Added `set_jit_options()`, `get_jit_options()`, and task/thread-local + `jit_options()` contexts. Configure JIT policy/backend, floating-point accuracy, + tracing, CC compiler/flags/cache directory and compiler output without changing + process environment variables. Explicit per-call settings remain authoritative + over Python defaults; existing environment overrides are preserved. +- Native compilation receives call-local settings. Toolchain settings now + participate in process-cache identity, preventing stale reuse after changing + compiler flags or the selected compiler. + +- Bundled TCC now uses anonymous memfd-backed RW/RX executable storage on Linux, + avoiding the executable-heap failure reported with SELinux (#730). Its runtime + path no longer creates or requires a filesystem JIT cache. No installed system + compiler or libselinux dependency is required. +- Native TCC/CC compilation and loading failures fall back quietly to miniexpr's + interpreter, including explicit `jit=True` requests. `ME_DSL_TRACE=1` reports + fallback reasons; compiler output remains opt-in. CC still supports optimized + compilation and persistent shared-library caching. +- Added a JIT options API reference, including constructor kwargs, backend + requirements, environment precedence, caching, and diagnostics. Corrected the + generated allocation declaration that caused an Apple Clang warning. + ## Changes from 4.14.0 to 4.14.1 Python-Blosc2 4.14.1 is a security and feature release introducing safe diff --git a/doc/conf.py b/doc/conf.py index cc467bd41..915bb0c20 100644 --- a/doc/conf.py +++ b/doc/conf.py @@ -212,6 +212,10 @@ def genbody(f, func_list, lib="blosc2"): Reduction operations can be used with any of :ref:`NDArray `, :ref:`C2Array `, :ref:`NDField ` and :ref:`LazyExpr `. Again, although these can be part of a :ref:`LazyExpr `, you must be aware that they are not lazy, but will be evaluated eagerly during the construction of a LazyExpr instance (this might change in the future). When the input is a :ref:`LazyExpr`, reductions accept ``fp_accuracy`` to control floating-point accuracy, and it is forwarded to :func:`LazyExpr.compute`. +Eligible lazy-expression reductions accept ``jit`` and ``jit_backend`` through +their evaluation options. See :ref:`JITOptions`; native JIT requests remain best +effort and can use the interpreter when compilation or loading is unavailable. + .. currentmodule:: blosc2 .. autosummary:: diff --git a/doc/reference/dsl_syntax.md b/doc/reference/dsl_syntax.md index cd4b5f8d7..2568d3171 100644 --- a/doc/reference/dsl_syntax.md +++ b/doc/reference/dsl_syntax.md @@ -33,6 +33,9 @@ explicit form — it always requires the DSL to compile, equivalent to - Leading blank lines and header comments are allowed. - Any extra trailing content after the function is a parse error. - Nested `def` inside the function body is not allowed. +- The first statement may be a docstring: single/double-quoted or triple-quoted, + including multiline strings and `r`/`u` prefixes. It is ignored during execution; + `@blosc2.dsl_kernel` preserves the Python function's `__doc__`. ## Header pragmas @@ -67,6 +70,7 @@ Supported statement forms: - While loop: `while cond:` - For loop: `for i in range(...):` - Loop control: `break`, `continue` +- No-op: `pass` (also valid as the sole statement in a branch or loop body) General rules: @@ -74,6 +78,28 @@ General rules: - Empty blocks are invalid. - `elif`/`else` must belong to a matching `if`. - Deprecated forms like `break if cond` / `continue if cond` are not part of DSL syntax. +- Simple statements may share a line, separated by `;`, including inside indented + blocks. A trailing `;` is allowed; empty statements (`;;`) are not. +- Compound statements (`if`, `for`, `while`) require their own lines and indented + bodies; they cannot follow a semicolon. Inline suites such as `if x: return x` + are not supported. +- Expressions and calls can continue across lines inside parentheses, including + comments. This does not add list literals or indexing to the expression grammar. + +### Docstrings and semicolon-separated statements + +```python +@blosc2.dsl_kernel +def kernel(x): + """Transform each element. + + Documentation is not evaluated by the interpreter or JIT. + """ + y = x + 1; z = y * y # fmt: skip + if z > 4: + z -= 2; z *= 3 # fmt: skip + return z +``` ### `if` / `elif` / `else` example @@ -114,6 +140,19 @@ def kernel(x): Expressions are compiled by miniexpr with DSL checks. +### Numeric literals + +Python-style digit separators are accepted in integers and floating-point +literals: `1_000`, `1_000.2_5`, `.1_25`, and `1e1_0`. Integer literals also accept +binary (`0b1010`), octal (`0o755`), and hexadecimal (`0xff`) prefixes, including +uppercase prefixes and separators such as `0x_FF` or `0b10_10`. + +Malformed separators, invalid base digits, and nonzero decimal integers with +leading zeros are rejected. Prefixed integer magnitudes must fit in an unsigned +64-bit value; numeric evaluation retains the existing dtype and precision rules, +not Python's arbitrary-precision integer arithmetic. Strings and identifiers +(such as `x_1`) are not modified. + Commonly supported: - Names and numeric constants @@ -123,6 +162,20 @@ Commonly supported: - Function calls to supported miniexpr functions - User-registered C functions/closures passed in `me_variable` +DSL kernels accept chained comparisons such as `0 <= x < 10` or +`a < b <= c != d`. Operands are evaluated left-to-right, each at most once; +later operands are skipped as soon as a comparison fails. Chains work in +expressions, assignments, branches, and loop conditions (including `continue`). +The native miniexpr DSL front end lowers them to temporary variables and guarded +statements for both the interpreter and TCC/CC JIT. C and other callers can pass +raw chain syntax directly, without Python preprocessing. JavaScript emission +performs equivalent lowering separately. Existing DSL `and`/`or` Boolean-result +rules apply. This syntax extension applies to DSL kernels, not the separate +classic expression API. + +Chaining does not expand the supported operand types or functions: existing +string-comparison and per-element reduction restrictions still apply. + Cast intrinsics: - `int(expr)` @@ -151,6 +204,8 @@ In this example, `temp` is inferred from `sin(x) ** 2` (typically a floating typ Notes: - You do not need to declare local variable types. +- Boolean output does not force numeric temporaries to Boolean: operand types + are inferred independently, and the return value is converted to Boolean. - If you assign a value with an incompatible dtype to the same local later, compilation fails. ## Loops @@ -230,14 +285,18 @@ When referenced, these are synthesized by DSL compiler/runtime: ### Compute dtype and integer exactness -The kernel's *output* dtype determines the compute dtype for the whole expression: +The kernel's *output* dtype generally determines the arithmetic compute dtype: - With an integer output dtype, arithmetic is exact int64. Intermediates must fit in int64: products at or above 2^63 overflow and give wrong results. -- With a float output dtype, integer inputs and temporaries are evaluated in float64, +- With a float output dtype, integer arithmetic is generally evaluated in float64, where integer operations are exact only below 2^53. Keep products under that bound (e.g. a 32-bit value times a multiplier below 2^21); larger products silently lose low bits. +- Native chained-comparison captures retain each operand's inferred dtype, so + comparisons of int64 operands can remain exact even with a float output. + Automatic JavaScript dispatch therefore leaves 64-bit integer input arrays on + miniexpr; the JavaScript bridge converts inputs to float64. - Values outside the output dtype's range wrap two's-complement on the final store (e.g. returning a value in `[0, 2^32)` into an int32 output yields the full `[-2^31, 2^31)` range). @@ -269,12 +328,19 @@ Runtime error examples: ## Execution backends -A DSL kernel is compiled and run by one of two backends, selected per evaluation -via the `jit` / `jit_backend` arguments to `compute()` / `__getitem__`: +A DSL kernel can use the following execution backends. Set `jit` / `jit_backend` +on `lazyudf()` or `compute()`; indexing uses the configured/default settings. +See the [JIT options reference](jit.rst) for defaults, environment precedence, +caching, and diagnostics: - **miniexpr** (default on native builds): a runtime JIT (TinyCC, `jit_backend="tcc"`) - with an interpreter fallback (`jit=False`). Supports the full DSL described here, + with an interpreter fallback. `jit=False` skips JIT; `jit=True` is best effort + and also falls back if executable allocation or compilation is denied. Supports the full DSL described here, including integer/complex dtypes and reductions. +- **System C compiler** (`jit_backend="cc"`): miniexpr generates optimized shared + libraries using an installed compiler and caches them for subsequent processes. + Compilation or loading failures also fall back to the interpreter. TCC instead + compiles in memory without creating JIT cache artifacts. - **JavaScript** (`jit_backend="js"`): transpiles the kernel to JavaScript and runs it through the browser's JIT. **WebAssembly/Pyodide only** — requesting it elsewhere raises. Under WebAssembly it is also the *default* for eligible kernels (set `jit=False` or diff --git a/doc/reference/index.rst b/doc/reference/index.rst index 9c5cb807f..a0ba44c75 100644 --- a/doc/reference/index.rst +++ b/doc/reference/index.rst @@ -12,3 +12,4 @@ API Reference save_load utilities dsl_syntax + jit diff --git a/doc/reference/jit.rst b/doc/reference/jit.rst new file mode 100644 index 000000000..ecae2de28 --- /dev/null +++ b/doc/reference/jit.rst @@ -0,0 +1,301 @@ +.. _JITOptions: + +JIT execution options +===================== + +Blosc2 can compile eligible computations at runtime (JIT compilation). Native +builds use the bundled Tiny C Compiler (TCC) by default, so installing a system +compiler is **not** necessary to use Blosc2 or its default JIT. + +Global defaults and scoped options +---------------------------------- + +Use :func:`blosc2.set_jit_options` to configure a script, or +:func:`blosc2.jit_options` for a temporary thread/task-local override: + +.. code-block:: python + + import blosc2 + + previous = blosc2.set_jit_options(jit_backend="cc") + a = blosc2.linspace(0, 10, 10_000) + expr = a * 2 + 1 + with blosc2.jit_options(jit=True, trace=True, cflags="-O2"): + result = expr.compute() + blosc2.set_jit_options(**previous) + +Both functions accept the same keyword-only parameters: + +.. list-table:: + :header-rows: 1 + :widths: 25 20 55 + + * - Parameter + - Built-in default + - Meaning + * - ``jit`` + - ``None`` + - Existing best-effort JIT policy; ``True`` also requests eligible plain + expression auto-lifting. ``False`` disables JIT. + * - ``jit_backend`` + - ``None`` + - ``"tcc"``, ``"cc"``, or ``"js"``. JavaScript is WebAssembly/Pyodide-only. + * - ``fp_accuracy`` + - ``FPAccuracy.DEFAULT`` + - Existing floating-point function accuracy policy, not a compiler flag. + Use an actual :class:`blosc2.FPAccuracy` member. + * - ``trace`` + - ``False`` + - Python routing and native compilation/cache/fallback diagnostics on + stderr. Does not enable external compiler output. + * - ``compiler`` + - ``None`` + - CC compiler command, normally ``cc``. A string or path-like object. + Quoted commands can name executables with spaces; path-like objects + are converted to shell-quoted executable paths automatically. + * - ``cflags`` + - ``None`` + - Additional CC flags as a string, appended to the normal optimization and + floating-point flags. Compiler-specific; changing them can affect accuracy. + * - ``cache_dir`` + - ``None`` + - Actual persistent CC cache directory, not a ``TMPDIR`` root. Accepts a + string/path-like object and resolves relative paths when configured. + Created on demand only by CC; TCC does not use it. + * - ``compiler_output`` + - ``False`` + - Opt into the external compiler's output independently of ``trace``. + +Compiler commands/flags are trusted shell configuration, not safe inputs from +untrusted users. Choose protected cache storage. CC-specific options have no +effect on TCC, JavaScript, or non-JIT execution. + +For setters and contexts, an omitted argument leaves/inherits that setting; +``None`` resets it to the built-in default. The setter returns the previous +process defaults as a fresh dictionary. :func:`blosc2.get_jit_options` returns +a fresh dictionary of the current Python defaults, including context overrides, +but excluding environment overrides and per-evaluation arguments. Validation is +atomic: an invalid setting leaves the defaults unchanged. + +Within Python, precedence is: + +**explicit evaluation settings → explicit LazyUDF settings → context defaults → +process defaults → built-in defaults**. + +The execution options also work as per-call kwargs on the evaluation entry points +listed below. For these existing per-call APIs, ``None`` means inherit, not reset; +use a resetting context to select built-in defaults locally. ``fp_accuracy`` now +defaults to ``None`` on lazy compute/reduction methods, so omitted accuracy +inherits rather than masking configured defaults. Without configured defaults, +behavior is unchanged. + +Settings resolve at evaluation time: an expression constructed outside a context +can be evaluated inside it. Explicit settings attached to a LazyUDF remain +authoritative unless overridden on its ``compute()`` call. Changes do not modify +already compiled kernels. Contexts nest and restore settings even after exceptions; +they never mutate environment variables or process-wide defaults. Async tasks +inherit their creation context but do not affect unrelated tasks. Thread context +propagation follows Python's ``contextvars`` rules (including explicit copied +contexts, ``asyncio.to_thread()``, and Python's thread-context inheritance setting). +Process-wide setters still affect all threads without a contextual override. + +Compiler command/flags and explicit cache directories participate in native cache +identity, including process-local positive/negative caches. Changing them must not +reuse an incompatible kernel. Tracing/compiler-output toggles do not change cache +identity. ``fp_accuracy`` retains the existing evaluator accuracy semantics; +it does not select miniexpr's separate strict/contract/fast compiler mode. + +Existing nonempty environment overrides remain authoritative: ``CC`` overrides +``compiler``, ``CFLAGS`` overrides ``cflags``, ``ME_DSL_TRACE`` overrides ``trace``, +``ME_DSL_JIT_DEBUG_CC`` overrides ``compiler_output``, and +``ME_DSL_JIT_CACHE_DIR`` overrides ``cache_dir``. ``TMPDIR`` is used only when no +explicit cache directory is selected. The JIT enable/backend environment rules +below remain unchanged. Configuration does not require changing environment +variables; native compilation receives a call-local settings snapshot. + +.. autofunction:: blosc2.set_jit_options + +.. autofunction:: blosc2.get_jit_options + +.. autofunction:: blosc2.jit_options + +JIT is best effort +------------------ + +``jit=True`` prefers JIT; it does not require successful native compilation. +If TCC or the explicitly selected system compiler cannot compile or load a +kernel, miniexpr evaluates it with its interpreter instead. This also applies +when SELinux or another operating-system policy denies executable mappings. +Fallback is quiet by default and may be slower. Invalid arguments, invalid DSL +syntax, unsupported backend names, and genuine computation errors still raise +their normal errors. + +Successful array creation alone is not evidence that JIT ran. Use +``ME_DSL_TRACE=1`` to inspect runtime compilation, cache reuse, and fallback, or +:func:`blosc2.validate_dsl_jit` to probe a DSL kernel for specific input/output +dtypes. The latter probes compilation, not execution on real data. + +Per-call controls +----------------- + +These options are accepted by :meth:`blosc2.LazyArray.compute` and +:func:`blosc2.lazyudf`. :func:`blosc2.arange` and :func:`blosc2.linspace` also +accept them through ``**kwargs``. Eligible lazy-expression reductions, such as +``expr.sum(jit=True, jit_backend="cc")``, forward them to evaluation. These +settings tune execution rather than array storage; they are not options to +:func:`blosc2.empty`. + +``jit=None`` (default) + Use the entry point's default policy. DSL kernels, including the real-valued + ramps used by ``arange`` and ``linspace``, try JIT. Plain expressions do not + automatically become DSL kernels unless JIT is requested. Ordinary Python + callback UDFs are not compiled by TCC/CC through this option. + +``jit=True`` + Prefer JIT for eligible computations. Plain expressions can be automatically + lifted into DSL kernels. Interpreter fallback remains available. + +``jit=False`` + Disable JIT for the computation. Applicable environment overrides are + described below. Depending on the expression, the non-JIT route can be the + miniexpr interpreter or another existing compute engine. + +``jit_backend=None`` (default) + Use the default backend: TCC on supported native targets. Under + WebAssembly/Pyodide, eligible floating-point DSL kernels prefer the JavaScript + backend unless ``jit=False`` or ``strict_miniexpr=True`` is specified. + +``jit_backend="tcc"`` + Select bundled TCC. It compiles generated C in memory, retains the compiled + state for evaluation, and does not persist binaries for another Python + process. It creates no JIT source, binary, metadata, or cache-directory files. + On Linux, executable storage uses anonymous memfd-backed RW/RX mappings; + it does not need a writable/executable temporary directory. Installed + libraries still need to be loadable normally. If the target lacks a TCC + backend, or allocation is denied, execution uses the interpreter. + +``jit_backend="cc"`` + Select an installed C compiler (normally GCC or Clang), with optimized code + generation. This needs a compiler and writable cache storage that permits + loading executable shared libraries. Compiled libraries and metadata persist + for reuse by subsequent processes. Compiler absence, compilation failure, + cache failure, or denied library loading falls back to the interpreter. + For a plain expression, also set ``jit=True`` to request DSL auto-lifting. + +``jit_backend="js"`` + Select the JavaScript bridge, available only under WebAssembly/Pyodide. + Explicit selection on native builds raises an error. Eligibility and explicit + unsupported-kernel errors are described in the + `DSL syntax reference `_. + +``fp_accuracy`` controls numerical accuracy independently of the backend. +TCC currently supports the strict miniexpr compiler floating-point mode; +unsupported JIT modes can use the interpreter. Backends need not produce +bit-identical floating-point results. ``strict_miniexpr`` controls whether +miniexpr compilation/evaluation failures are raised instead of changing compute +engines; it is **not** a requirement that native JIT succeed. A valid kernel may +still use miniexpr's interpreter. + +Examples +-------- + +.. code-block:: python + + import blosc2 + + a = blosc2.linspace(0, 10, 10_000) # Default: try bundled TCC + a = blosc2.linspace(0, 10, 10_000, jit=False) + a = blosc2.arange(10_000, jit=True, jit_backend="tcc") + a = blosc2.linspace(0, 10, 10_000, jit=True, jit_backend="cc") + + expr = a * 2 + 1 + result = expr.compute(jit=True, jit_backend="cc") # NDArray + + + @blosc2.dsl_kernel + def squared(x): + return x * x + + + udf = blosc2.lazyudf(squared, (a,), dtype=a.dtype, jit_backend="tcc") + values = udf[:] # NumPy values + +Environment settings and precedence +----------------------------------- + +Set environment variables before starting Python. Python-level and miniexpr-level +controls have different scopes; neither is an indication that JIT actually ran. + +.. list-table:: + :header-rows: 1 + :widths: 32 68 + + * - Variable + - Meaning + * - ``BLOSC_ME_JIT`` + - On ``LazyExpr.compute()`` (and paths calling it), values ``1``, ``true``, + or ``on`` override ``jit`` to ``True`` without changing a supplied backend. + Values ``tcc`` and ``cc`` additionally override ``jit_backend``. + This is not a universal constructor/``LazyUDF`` override. ``0`` does not + disable JIT: use per-call ``jit=False`` or ``ME_DSL_JIT=0`` instead. + * - ``ME_DSL_JIT=0`` + - Disable miniexpr runtime JIT, including when a Python call requests + ``jit=True``. This does not control the separate JavaScript backend. + * - ``ME_DSL_TRACE=1`` + - Print native code-generation/runtime diagnostics to stderr, including + builds, cache hits, failure explanations, and interpreter fallback. + * - ``BLOSC_ME_JIT_TRACE=1`` + - Print Python compute-engine routing on supported paths to stdout. + Distinct from native JIT success reporting. + * - ``ME_DSL_JIT_DEBUG_CC=1`` + - Show the external compiler's normally suppressed output. + * - ``CC`` and ``CFLAGS`` + - Select and configure the system compiler backend; defaults to ``cc`` + with optimized compilation. Not needed by TCC. + * - ``TMPDIR`` + - CC cache root: artifacts go into ``$TMPDIR/miniexpr-jit``. When unset, + Linux/macOS use ``/tmp/miniexpr-jit-``. TCC does not use this cache. + * - ``ME_DSL_JIT_CACHE_DIR`` + - Exact CC cache directory; overrides Python ``cache_dir`` and the + ``TMPDIR``-based default. TCC ignores it. + * - ``ME_DSL_JIT_TCC_OPTIONS`` + - Extra options for compiling generated C with TCC. Not build flags that + change libtcc's executable allocator. + * - ``ME_DSL_JIT_LIBTCC_PATH`` + - Advanced: explicitly override the loaded libtcc path. A failed override + does not silently substitute another library; evaluation can fall back. + * - ``ME_DSL_JIT_POS_CACHE=0`` + - Disable applicable process-local positive-cache reuse. This does not + disable persistent CC disk-cache reuse. + * - ``ME_DSL_JIT_COMPILER`` + - Advanced miniexpr-wide compiler override, ``tcc`` or ``cc``. It overrides + native compiler selection in DSL pragmas, including the selection emitted + by Python's ``jit_backend`` option. Prefer per-call options ordinarily. + +Tracing and restricted environments +----------------------------------- + +.. code-block:: sh + + env ME_DSL_TRACE=1 python -c \ + 'import blosc2; blosc2.linspace(0, 10, 10_000, jit=True, jit_backend="tcc")' + +Representative runtime messages include: + +.. code-block:: text + + [me-dsl] jit runtime built: fp=strict compiler=tcc key=... + [me-dsl] jit runtime hit: fp=strict source=disk-cache key=... + [me-dsl] jit runtime fallback: interpreter fp=strict compiler=tcc reason=... + +An executable-memfd denial can disable TCC JIT. A ``noexec`` cache mount or an +SELinux file-execution restriction can disable CC JIT. Neither normally prevents +a valid computation from running with the interpreter. No system-policy changes +are required to use that fallback. + +A fresh Python process is not necessarily a cache-cold CC run: it can reuse an +earlier shared library. Use a fresh ``TMPDIR`` when measuring CC compilation. +Measure Python startup, compilation/linking, first library loading, and array +execution separately when comparing backends. TCC prioritizes startup latency; +CC often improves throughput on large computations. Writing a persistent user +array with ``urlpath=`` is independent of JIT artifact storage. diff --git a/doc/reference/lazyarray.rst b/doc/reference/lazyarray.rst index d77844c59..bd6368a71 100644 --- a/doc/reference/lazyarray.rst +++ b/doc/reference/lazyarray.rst @@ -21,6 +21,9 @@ underlying carrier and survives reopening. See the `LazyExpr`_ and `LazyUDF`_ sections for more information. +See :ref:`JITOptions` for ``jit``/``jit_backend`` controls, environment settings, +backend caching, and automatic interpreter fallback. + .. currentmodule:: blosc2 .. autoclass:: LazyArray diff --git a/doc/reference/ndarray.rst b/doc/reference/ndarray.rst index 555678217..30d07cd5c 100644 --- a/doc/reference/ndarray.rst +++ b/doc/reference/ndarray.rst @@ -38,6 +38,10 @@ It is a direct alias for ``array.vlmeta`` and uses the same persistent storage. Constructors ------------ +The ``arange`` and ``linspace`` constructors accept ``jit`` and ``jit_backend`` +execution options through ``**kwargs``. See :ref:`JITOptions` for defaults, +backend requirements, and quiet interpreter fallback. + .. _NDArrayConstructors: .. autosummary:: diff --git a/doc/reference/reduction_functions.rst b/doc/reference/reduction_functions.rst index 88f19146b..564680242 100644 --- a/doc/reference/reduction_functions.rst +++ b/doc/reference/reduction_functions.rst @@ -5,6 +5,10 @@ Contrarily to lazy functions, reduction functions are evaluated eagerly, and the Reduction operations can be used with any of :ref:`NDArray `, :ref:`C2Array `, :ref:`NDField ` and :ref:`LazyExpr `. Again, although these can be part of a :ref:`LazyExpr `, you must be aware that they are not lazy, but will be evaluated eagerly during the construction of a LazyExpr instance (this might change in the future). When the input is a :ref:`LazyExpr`, reductions accept ``fp_accuracy`` to control floating-point accuracy, and it is forwarded to :func:`LazyExpr.compute`. +Eligible lazy-expression reductions accept ``jit`` and ``jit_backend`` through +their evaluation options. See :ref:`JITOptions`; native JIT requests remain best +effort and can use the interpreter when compilation or loading is unavailable. + .. currentmodule:: blosc2 .. autosummary:: diff --git a/plans/portable-DSLs.md b/plans/portable-DSLs.md new file mode 100644 index 000000000..f0d2ffc33 --- /dev/null +++ b/plans/portable-DSLs.md @@ -0,0 +1,184 @@ +# Portable DSL kernels and language-specific authoring frontends + +**Status:** Proposal for later consideration; no implementation or API commitment. + +## Motivation + +Python authoring conveniences make DSL kernels pleasant to write, but their source +may depend on Python-specific rewriting, captured globals, or pandas conventions. +When other languages support Blosc2 lazy arrays, they should not need Python to +load or execute a kernel authored in Python. + +The guiding principle is **portable execution artifacts, flexible authoring +frontends**. Separate the canonical DSL understood by miniexpr from the authoring +dialects offered by Python and future Rust, Julia, JavaScript, or other frontends. + +## Current boundary + +Chained comparisons now belong to the native miniexpr DSL front end. C callers can +pass their raw syntax inside DSL kernels, with single operand evaluation and +short-circuiting. Docstrings, semicolons, multiline expressions/comments, `pass`, +and numeric literal conveniences are also native syntax. + +Five convenience families still depend on Python rewriting: + +| Python authoring syntax | Canonical native representation | +| --- | --- | +| String methods such as `s.lower()` | `lower(s)` | +| String membership: `x in s`, `x not in s` | `contains(s, x)`, optionally negated | +| Two-part split unpacking: `a, b = s.split(sep, 1)` | Two `split_part(...)` assignments | +| Pandas row access: `row["column"]` | Named kernel parameters and a column mapping | +| NumPy calls such as `np.sin(x)` and `np.maximum(a, b)` | Native calls such as `sin(x)` and `fmax(a, b)` | + +Python also specializes captured scalar constants. This is another preprocessing +dependency, rather than a syntax feature. JavaScript emission has its own lowering +for comparison chains; that is a backend adapter, not a Python dependency of native +execution. + +## Proposed architecture + +### 1. Specify a canonical, language-independent DSL + +Use native miniexpr DSL syntax and function semantics as the portable contract. +Document the boundary explicitly: an authoring frontend may accept more syntax, +but exported artifacts must conform to the canonical language. + +Specify evaluation order, short-circuiting, dtype conversions, numeric precision, +string/bytes behavior, and reduction restrictions. Portability must not imply +Python's arbitrary-precision integers or promise identical floating-point bits +across backends and accuracy modes. + +Keep the DSL kernel language distinct from miniexpr's classic expression API. + +### 2. Normalize at export time, not at load time + +Provide a future export step that resolves frontend-specific syntax and implicit +dependencies into a portable kernel. A foreign-language loader must not import +Python, reconstruct a Python function, inspect its globals, or execute frontend +rewriters. + +Normalization must preserve supported semantics, including operand evaluation +count and order. For example, lowering split unpacking must not evaluate a +nontrivial subject or separator twice merely because two parts are extracted. +Unsupported or ambiguous constructs should fail export with actionable errors. + +The existing `kernel.dsl_source` is useful source material, but should not be +treated as a complete portable artifact: bindings and frontend metadata may still +be needed. + +### 3. Make the portable kernel self-describing + +An artifact should contain, or explicitly reference: + +- Canonical DSL source and its entry point. +- Ordered input names and language-neutral dtype descriptions. +- Output type and any required shape, rank, or index-symbol constraints. +- Captured scalar constants as explicit, typed bindings. +- Any column-to-parameter mapping, without requiring pandas objects. +- A DSL language version, an artifact schema version, and required capabilities. +- Semantic execution requirements, such as floating-point accuracy expectations. + +Separate required semantics from optional execution preferences. Compiler paths, +cache directories, diagnostic toggles, and host-specific JIT settings should remain +local runtime configuration rather than requirements embedded in portable kernels. + +The container format is intentionally undecided. Evaluate a standalone manifest +and integration with existing Blosc2 persistence rather than introducing a new +storage format prematurely. Do not store generated machine code as the portable +representation. + +Typed bindings need a defined encoding for integer widths, floating-point values +(including non-finite values), strings, and bytes. Do not rely on Python `repr`, +NumPy dtype objects, or pickle as the cross-language contract. + +### 4. Offer an optional strict portability mode + +A future authoring/validation option could reject frontend-only syntax and implicit +captures rather than normalize them. For example: + +> Use `lower(s)` instead of `s.lower()` for portable DSL source. + +Strict mode would help users write source that can be copied directly into another +language's runtime. It should be opt-in; existing convenient Python authoring must +remain supported. + +No particular option name or API signature is proposed yet. + +### 5. Validate independently of Python + +Treat export success and target compatibility as separate checks. An artifact can +be valid canonical DSL while requiring capabilities absent from a particular +runtime or backend. + +Validate schema, language version, input bindings, and required capabilities before +execution. Report unsupported requirements explicitly. Where an interpreter can +honor the required semantics, lack of native JIT should retain the existing +best-effort fallback policy. + +External C functions or closures need explicit handling: either reject them for a +self-contained export or describe stable symbolic function requirements that a +loader must register. Never serialize process-local pointers or Python closures. + +Portable does not mean safe to execute from an untrusted source. Define resource +limits and trust requirements separately; an artifact must not authorize loading +arbitrary libraries, running shell commands, or executing Python during loading. + +## Which conveniences should become native? + +- **NumPy aliases and pandas row access:** keep frontend-only. They are ecosystem + integrations, not necessary parts of the language-independent DSL. +- **String methods, membership, and split unpacking:** possible future native + syntax extensions, but not required for portability if export normalizes them. + Decide based on usefulness across languages and maintenance cost. +- **Captured constants:** represent them as explicit bindings in the artifact. +- **General language constructs:** consider native implementation when they carry + broadly useful semantics, as with chained comparisons. + +Avoid moving conveniences into miniexpr solely to eliminate Python rewrites. The +goal is a stable portable contract, not reproduction of the entire Python surface. + +## Phased work + +1. **Inventory and specification:** document native syntax, frontend extensions, + implicit dependencies, and the initial portable capability set. +2. **Artifact design:** agree on versions, schemas, typed bindings, and whether + portable kernels are standalone or attached to persisted lazy arrays. +3. **Python export prototype:** normalize conveniences and resolve scalar captures + without changing existing evaluation APIs. Add optional strict validation. +4. **Native loading and execution:** implement a C-facing validation/loading path + that accepts the artifact without Python preparation. +5. **Cross-language conformance:** establish a reusable corpus before adding further + language frontends and backend capability profiles. + +Exporting an entire persisted lazy computation graph is a separate follow-up. A +portable kernel contract alone does not specify array references, storage access, +graph execution, or remote authentication. + +## Validation and acceptance criteria + +- Export from Python, then load and execute from a standalone C process without + importing Python or applying Python-side preparation. +- Cover all five convenience families, explicit scalar bindings, column mappings, + and kernels already written in canonical syntax. +- Check evaluation order, single evaluation, short-circuiting, numeric boundaries, + string/bytes behavior, and documented reduction rules. +- Compare interpreter, TCC/CC, and applicable JavaScript results under declared + capability and accuracy profiles; do not require unavailable backend features. +- Reject unresolved globals, malformed bindings, unsupported versions or + capabilities, missing external functions, and unsafe loading instructions. +- Verify that frontend implementation changes do not break previously exported + artifacts within the promised compatibility policy. +- Preserve current Python authoring and quiet best-effort JIT fallback behavior. + +## Open decisions + +- Minimum initial dtype/function capability set and capability naming. +- Canonical source normalization and whether source maps accompany diagnostics. +- Artifact encoding, storage location, and typed scalar representation. +- Version compatibility and migration policy. +- Whether runtime shape constraints are fixed or parameterized. +- Registration and versioning of optional external functions. +- Semantic compatibility rules for floating-point accuracy and backend differences. + +These decisions should be reviewed before implementation; this plan records the +direction without expanding the current DSL API or persistence contract. diff --git a/plans/safer-jit-execution.md b/plans/safer-jit-execution.md new file mode 100644 index 000000000..7114c1802 --- /dev/null +++ b/plans/safer-jit-execution.md @@ -0,0 +1,447 @@ +# Safer, filesystem-independent TCC JIT execution + +Status: implementation and macOS integration validation completed; Linux policy +and platform validation remain open. See [the review](safer-jit-review.md) for +test results, fixes, remaining weak points and release gates. Checklists below +retain the original validation scope and are not a claim of full platform coverage. + +Issue: [Support for SELinux, python-blosc2 #730](https://github.com/Blosc/python-blosc2/issues/730). + +## Objective + +Make the default JIT work transparently on Linux installations with SELinux, +without requiring a system C compiler or a writable/executable temporary +directory. When the operating system denies JIT execution, valid computations +should continue through the miniexpr interpreter, quietly by default. + +Keep bundled TCC as the default native compiler. Keep `cc` explicitly selectable +for optimized execution and persistent shared-library caching. The measurements +behind this decision show that compiling a simple kernel can take only about +30 ms on Linux, but that is still a meaningful fixed cost for small operations. +On macOS, first loading a newly generated library can cost substantially more +than compiling it. Backend selection should remain predictable rather than +introducing workload thresholds as part of this work. + +## Agreed execution contract + +| Situation | Expected behavior | +| --- | --- | +| Default native JIT, supported target | Try bundled TCC; no system compiler required | +| Linux TCC executable allocation | Use anonymous memfd-backed RW/RX mappings | +| TCC compilation and execution | Keep generated code and its bookkeeping in memory | +| TCC unavailable or executable allocation denied | Clean up and use the interpreter | +| Explicit `jit_backend="cc"` | Try the system compiler and existing persistent cache | +| CC unavailable, compilation fails, or library loading is denied | Use the interpreter | +| `jit=False`, absent an applicable environment override | Skip JIT and use the existing non-JIT route | +| `jit=True` | Prefer JIT; successful JIT compilation is not required | +| Native TCC/CC backend explicitly selected | Select the attempted backend; retain interpreter fallback | +| Ordinary execution without debug settings | No raw compiler diagnostics or fallback warnings | +| Opt-in tracing | Report the attempted backend, failure reason, and interpreter fallback | +| Invalid expressions, arguments, or unsupported backend names | Preserve normal validation errors | + +The filesystem contract concerns JIT artifacts: TCC must neither create nor +require source files, binaries, metadata files, or cache directories. Normal +loading of installed libraries and OS paging remain ordinary platform behavior. +Persistent user arrays requested with `urlpath=` are independent of this contract. + +Fallback preserves the supported computation and its numerical-accuracy +contract; it may be slower. It does not promise identical floating-point bit +patterns across all backends or successful execution after unrelated failures. +Native macOS and Windows retain their platform-specific in-memory allocation +mechanisms. WebAssembly retains its existing backend selection and helpers. + +## Starting points in the code + +Paths prefixed with `minicc/` and `miniexpr/` refer to those separate repositories; +they are available locally as `../minicc` and `../miniexpr`. + +| Area | Main files | +| --- | --- | +| TCC runtime allocation, relocation, and release | `minicc/tccrun.c`, `minicc/tcc.h` | +| TCC build configuration | `minicc/CMakeLists.txt`, `minicc/cmake/config.h.in`, `minicc/configure` | +| Bundled TCC build and revision | `miniexpr/cmake/MiniexprTinyccTargets.cmake`, `miniexpr/cmake/MiniexprFetchDeps.cmake` | +| Dynamic libtcc API and compilation | `miniexpr/src/dsl_jit_backend_libtcc.c` | +| Native JIT dispatch and caches | `miniexpr/src/dsl_jit_runtime_host.c`, `miniexpr/src/dsl_jit_runtime_cache.c` | +| Other platform dispatch | `miniexpr/src/dsl_jit_runtime_nonhost.c` | +| System compiler and shared-library loading | `miniexpr/src/dsl_jit_backend_cc.c` | +| Diagnostics and generated C | `miniexpr/src/dsl_config.h`, `miniexpr/src/dsl_jit_cgen.c` | +| Python dispatch and constructors | `src/blosc2/lazyexpr.py`, `src/blosc2/ndarray.py`, `src/blosc2/blosc2_ext.pyx` | +| Python DSL validation helpers | `src/blosc2/dsl_kernel.py` | +| Bundled miniexpr revision | `CMakeLists.txt` | + +Current behavior that motivates the changes: + +- Linux TCC normally uses malloc-backed storage and changes code-page permissions + with `mprotect()`. The reported SELinux denial occurs on that path. +- The optional `CONFIG_SELINUX` path uses a hardcoded `/tmp/.tccrunXXXXXX` backing + file. Its creation/resizing and partial mapping failures need better handling. +- Miniexpr builds minicc through CMake, rather than its `configure` script. + Merely supplying `--with-selinux` elsewhere does not configure the wheel's libtcc. +- TCC already compiles in memory and retains its `TCCState`; it does not write + reusable compiled binaries. However, host dispatch prepares and probes the disk + cache before reaching the TCC branch, introducing an unnecessary dependency. +- A relocation failure is already checked and normally permits interpreter + fallback. The missing libtcc diagnostic callback allows raw stderr messages to + make a recoverable failure look fatal. +- CC builds position-independent shared libraries and loads them with `dlopen()`. + SELinux file permissions or a `noexec` cache mount can still deny that route. + +## Task A: implement Linux memfd-backed allocation in minicc + +### A1. Make the build select the intended allocator + +- [ ] Make memfd-backed executable allocation the default for native Linux TCC + builds, including the shared libtcc shipped by miniexpr. +- [ ] Use target-platform configuration, including cross-compilation settings; + selection must not depend on whether SELinux is enabled on the build machine. +- [ ] Wire the allocator selection through CMake and keep the supported configure + build consistent. Ensure legacy `CONFIG_SELINUX` handling cannot accidentally + select the temporary-file allocator for the bundled Linux configuration. +- [ ] Keep the implementation self-contained: no libselinux dependency or runtime + SELinux detection is needed. Allocation success determines availability. +- [ ] Support the wheel's libc baseline. Where necessary, use a guarded syscall + wrapper instead of requiring a newer `memfd_create` libc symbol or newer headers. +- [ ] Forward any required build setting through miniexpr's minicc sub-build. + +### A2. Allocate one backing object and two aliases + +- [ ] Create an anonymous memfd with close-on-exec semantics. Request executable + memfd capability explicitly on kernels supporting the relevant flags. +- [ ] Handle older kernels deliberately: an unsupported new flag may justify + retrying with the compatible flag set. An actual permission denial must remain + a denial. Distinguish `EINVAL`, `ENOSYS`, `EACCES`, and `EPERM` where possible. +- [ ] Validate page alignment, nonzero allocation sizes, multiplication overflow, + and the representable range of TCC's existing size/offset fields. +- [ ] Check `ftruncate()` before mapping or accessing the object. A successful + mapping beyond the backing object's real length can otherwise lead to SIGBUS. +- [ ] Reserve a contiguous virtual-address region large enough for both aliases, + then establish an RX mapping and an RW mapping of the same memfd contents. + Check the reservation before using `MAP_FIXED` within that owned region. +- [ ] Preserve TCC's relocation model: executable symbols refer to the RX alias, + emitted code is copied through the RW alias, and writable data/BSS use their + intended writable addresses. Preserve the required fixed-distance offset. +- [ ] Close the descriptor after the mappings have been established. The mappings + keep the backing object alive without a filesystem name. +- [ ] Retain the required mappings for the compiled state's lifetime. In + particular, do not unmap the entire RW alias after code generation: TCC may + still need that alias for mutable data. + +No individual mapping should request simultaneous write and execute permission. +Separate aliases do not guarantee that every security policy permits execution; +mapping failure remains a normal reason to use the interpreter. + +### A3. Finish code publication and handle every failure + +- [ ] Perform architecture-correct instruction-cache synchronization before + exposing executable entry points, accounting for the RW and RX aliases. + The existing SELinux branch skips the helper where the ordinary path performs + ARM/AArch64 cache maintenance, so this needs explicit attention. +- [ ] Publish `run_ptr`, allocation size, and ownership metadata only after the + allocation is complete. Track enough information for deterministic cleanup. +- [ ] Use a single, complete failure path for descriptors, reservations, and + partially installed mappings. Preserve the original operation and `errno` + before cleanup can overwrite them. +- [ ] Report executable-allocation failures through a recoverable libtcc error + result. Audit this path for `exit()`/`abort()` calls; installing a diagnostic + callback alone does not make a fatal path recoverable. +- [ ] Update `tcc_run_free()` to release the allocation correctly on success and + on failures later in relocation or symbol lookup. +- [ ] On unsupported or denied memfd allocation, return failure directly. The + default Linux path must not create a temporary file or launch another compiler. + +## Task B: capture JIT diagnostics and preserve quiet fallback in miniexpr + +- [ ] Load and register `tcc_set_error_func()` alongside the other libtcc API + symbols, immediately after creating a state and before options, library lookup, + compilation, or relocation can produce diagnostics. +- [ ] Capture messages in a bounded diagnostic buffer owned by the appropriate + compilation/state. Do not redirect process-wide stderr. +- [ ] Keep callback data valid for the entire period in which libtcc may use it, + including state cleanup. A successful compilation must not retain a pointer to + an expired stack-local diagnostic buffer. +- [ ] Preserve useful details in the miniexpr error record: backend, operation or + phase, and the allocator/compiler/loader explanation. Avoid replacing a useful + diagnostic with only `tcc_relocate failed` or `compilation failed`. +- [ ] Leave the runtime kernel pointer unset on failure and release the failed + TCC state. Do not retry relocation on an already partially relocated state. +- [ ] Emit captured diagnostics only through opt-in tracing. Include an explicit + interpreter-fallback indication, rather than requiring users to infer it from + a generic JIT-skip message. Existing validation helpers should retain access to + an intelligible failure reason. +- [ ] Keep the same recoverable behavior for CC compiler absence, nonzero compiler + exit, cache access failures, and `dlopen()` failure. Compiler output remains + opt-in through `ME_DSL_JIT_DEBUG_CC`. +- [ ] Preserve validation and interpreter exceptions. Backend unavailability is + recoverable; invalid Python/DSL input must still be diagnosed normally. +- [ ] Follow existing locking and state-ownership conventions. Avoid introducing + races or cross-talk through shared diagnostic buffers during concurrent use. +- [ ] Preserve per-kernel negative caching. Add bounded backend/allocator-level + suppression for reliably identified environmental failures if needed to avoid + retrying the same denied operation for every new expression. Transient resource + errors and kernel-specific compilation errors must not permanently disable JIT. + Use structured status where available rather than parsing human-readable text. + +The public meaning of `jit=True` and explicit TCC/CC selection remains best effort. +Neither requests a new strict "JIT must succeed" execution mode. + +## Task C: remove TCC's dependency on filesystem caching + +- [ ] Refactor `dsl_try_prepare_jit_runtime()` so the native TCC branch runs before + all CC-specific disk-cache preparation and probing. +- [ ] Keep common eligibility checks, the runtime-disable setting, and applicable + in-memory negative-cache handling ahead of backend dispatch. +- [ ] Ensure the TCC branch does not call `dsl_jit_get_cache_dir()`, build artifact + paths, inspect disk metadata, read cached shared libraries, or create directories. +- [ ] Retain successful TCC states for their compiled program's lifetime and + continue releasing them through the normal program cleanup path. +- [ ] Keep CC's existing cache keys, metadata validation, shared-library loading, + process-local handle reuse, and persistent artifacts on the CC route. +- [ ] Ensure a denied TCC attempt proceeds directly to the interpreter, including + when a system compiler happens to be installed. +- [ ] Apply consistent diagnostic handling to the existing Windows/nonhost TCC + dispatcher without introducing host filesystem caching into that path. +- [ ] Confirm the macOS TCC path also bypasses disk-cache setup; memfd allocation + itself is Linux-specific. + +## Task D: integrate dependencies and verify Python dispatch + +- [ ] Add the fixed minicc revision to `miniexpr/cmake/MiniexprFetchDeps.cmake` + once available, and verify the sub-build produces the intended libtcc artifact. +- [ ] Add the updated miniexpr revision to Python-Blosc2's `CMakeLists.txt`. + Use existing local source overrides for development across the repositories. +- [ ] Exercise `arange`, `linspace`, lazy expressions, DSL `lazyudf`, and supported + reductions through their real Python/Cython dispatch paths. +- [ ] Confirm constructor `jit` and `jit_backend` kwargs reach evaluation rather + than storage constructors. Validate both omitted settings and explicit settings. +- [ ] Verify normal Python operations do not raise or issue warnings solely + because the requested native JIT backend is unavailable. +- [ ] Confirm wheels contain the updated libtcc and introduce no new system + compiler, libselinux, or newer-libc requirement. +- [ ] Add a release-note entry linking #730 and describing filesystem-independent + TCC execution and quiet interpreter fallback. + +## Task E: correct the generated allocation declaration + +The investigation exposed an Apple Clang warning for generated +`extern void *malloc(unsigned long long)`: that target's `size_t` is +`unsigned long`. + +- [ ] Replace the hardcoded declaration in miniexpr's C generator with one using + the actual target size type. Use compiler-provided type information or the + established target-aware generation mechanism. +- [ ] Preserve header-independent TCC compilation and verify LP64, LLP64, and + wasm32 type choices. Matching width alone does not make C prototypes compatible. +- [ ] Check related generated allocation declarations and casts for the same + narrow issue, and advance the code-generation cache version if required. +- [ ] Verify generated code compiles without the reported warning under Clang and + GCC, and still compiles through bundled TCC. + +## Task F: document JIT options in the API reference + +This is a required deliverable of #730. Users should be able to discover JIT +controls from the constructor and computation reference pages, including options +currently accepted through `**kwargs`. + +### F1. Add a canonical reference page and entry-point links + +- [ ] Add `doc/reference/jit.rst` with a stable cross-reference label and include + it in `doc/reference/index.rst`. +- [ ] Link it from `doc/reference/ndarray.rst`, `doc/reference/lazyarray.rst`, + `doc/reference/dsl_syntax.md`, and applicable reduction documentation. +- [ ] Explicitly document `jit` and `jit_backend` in the `arange()` and + `linspace()` docstrings in `src/blosc2/ndarray.py`; their `**kwargs` descriptions + must not imply that only `empty()` storage parameters are accepted. +- [ ] Audit and align the `LazyArray`/`LazyExpr`/`LazyUDF` compute and `lazyudf()` + documentation, plus relevant reduction entry points, with their actual support. +- [ ] Correct the DSL backend overview to include CC and automatic native + interpreter fallback. Show backend settings on supported constructors or + `compute()` calls; do not suggest passing arbitrary JIT kwargs to `__getitem__`. + +### F2. Document API semantics precisely + +The reference must cover: + +- `jit=None`: the API's default policy. Distinguish DSL kernels and constructors + implemented using them from plain expressions; do not promise every default + expression evaluation is JIT-compiled. +- `jit=True`: request/prefer JIT, including eligible plain-expression auto-lifting; + interpreter fallback remains available when compilation cannot be used. +- `jit=False`: request the supported non-JIT route, subject to documented + environment overrides on the entry points that implement them. +- `jit_backend=None` and `"tcc"`: bundled compiler, low startup cost, no persistent + TCC binary cache, and Linux memfd-backed execution without JIT artifact files. +- `jit_backend="cc"`: system compiler and optimized generated code, persistent + shared-library caching, filesystem requirements, and best-effort fallback. +- `jit_backend="js"`: existing WebAssembly/Pyodide-only support and eligibility + rules. Native selection of this backend remains an API error. +- The distinction between backend selection, `fp_accuracy`, and + `strict_miniexpr`. The latter is not a "require successful native JIT" flag. +- Current backend/target limitations, including TCC floating-point-mode support + and targets that use the interpreter because a native compiler backend is absent. +- Normal fallback is quiet and may reduce performance. Successful computation by + itself is not proof that JIT ran; use native tracing or `validate_dsl_jit()` on a + supported DSL kernel to inspect availability. + +### F3. Document environment controls and their scope + +Audit actual call paths and value parsing before writing the final table. In +particular, `_jit_from_env()` currently applies `BLOSC_ME_JIT` through +`LazyExpr.compute()`; do not describe it as a universal constructor override. +Document established precedence accurately and add focused checks where settings +interact. Any consistency fixes must be explicit, tested behavior changes. + +| Control | Required reference explanation | +| --- | --- | +| `BLOSC_ME_JIT` | Recognized enabling values and backend values; entry-point scope and precedence over supported per-call settings | +| `ME_DSL_JIT=0` | Disable miniexpr runtime JIT, including when Python requested it | +| `ME_DSL_TRACE=1` | Native codegen/runtime diagnostics, actual builds, cache hits, and fallback reasons on stderr | +| `BLOSC_ME_JIT_TRACE=1` | Python compute-engine routing on supported paths; distinct from native JIT success reporting | +| `ME_DSL_JIT_DEBUG_CC=1` | Opt into the external compiler's output | +| `CC`, `CFLAGS` | Select/configure the explicitly requested system compiler backend | +| `TMPDIR` | CC cache root and its default per-user location; no TCC artifact/cache dependency after this work | +| `ME_DSL_JIT_TCC_OPTIONS` | Options for generated-code compilation, not build flags that reconfigure libtcc's allocator | +| `ME_DSL_JIT_POS_CACHE` | Applicable process-local cache control; disabling it does not disable persistent CC cache reuse | + +Do not document `BLOSC_ME_JIT=0` as a global disabling switch: its current parser +does not implement that meaning. Include other miniexpr-specific controls only +after verifying their scope, precedence, and supported status. + +### F4. Add concise examples and troubleshooting guidance + +- [ ] Show default constructor use, `jit=False`, explicit TCC, and explicit CC: + + ```python + a = blosc2.linspace(0, 10, 10_000) + a = blosc2.linspace(0, 10, 10_000, jit=False) + a = blosc2.linspace(0, 10, 10_000, jit=True, jit_backend="tcc") + a = blosc2.linspace(0, 10, 10_000, jit=True, jit_backend="cc") + ``` + +- [ ] Show a plain lazy expression using `compute(jit=True)` and a DSL kernel + configured through `lazyudf()`. Follow the repository's result-type conventions: + `expr[:]` for values and `expr.compute()` for an `NDArray`. +- [ ] Show one trace-enabled shell command and representative messages for a + successful TCC build, a CC disk-cache hit, and interpreter fallback. +- [ ] Explain that a fresh Python process can reuse a CC binary. Cache-cold + benchmarking needs a fresh CC cache directory, and compilation, first library + loading, Python startup, and array execution are separate costs. +- [ ] Explain SELinux and `noexec` outcomes in terms of supported execution and + fallback. The ordinary solution must not require users to change system policy. +- [ ] Build the reference docs, check cross-references and generated signatures, + and execute the new examples in the supported environment. + +## Validation plan + +### 1. Native allocator tests in minicc + +Extend the existing libtcc API test coverage and its build integration. Use +test-local fault injection rather than adding public runtime failure switches. + +- Compile, relocate, and execute a small function; verify its result and mutable + global/data behavior through the RX/RW layout. +- Repeat allocation, execution, and destruction to detect descriptor and mapping + leaks. Exercise independent states and existing multithreaded test coverage. +- Inject failure at memfd creation, resize, reservation, RX mapping, and RW + mapping. Check recoverable status, original diagnostics, and complete cleanup. +- Test unsupported executable-memfd flags versus genuine permission denial, and + an unavailable syscall. Verify that no ordinary temporary file is created. +- Exercise boundary/overflow checks without attempting enormous real allocations. +- Verify executable mappings are not RWX. Validate ARM64 code publication on a + suitable Linux runner, rather than relying only on x86 cache coherence. + +### 2. Backend and fallback tests in miniexpr + +Extend `tests/test_dsl_jit_runtime_cache.c`, relevant DSL tests, and focused +libtcc-backend tests as needed. + +- TCC succeeds with an unusable `TMPDIR` and without a system compiler on `PATH`. + A regular file used as `TMPDIR` is a useful deterministic invalid-directory case. +- Instrument filesystem-cache helpers to confirm TCC neither creates nor probes + cache artifacts, including when a directory containing old kernels exists. +- Capture default stdout/stderr on denied TCC relocation and denied CC loading; + verify correct interpreter results and no unsolicited diagnostics. +- Enable tracing in a fresh process and verify backend, phase/reason, and explicit + fallback reporting. Test the external compiler's debug-output opt-in separately. +- Verify `jit=False` makes no native JIT attempt and `jit=True` permits fallback. +- Test compiler absence, failed compilation, and failed loading independently for + CC. Preserve successful cold-build and cross-process disk-cache reuse tests. +- Verify failed states and callback contexts are released; test bounded retry + behavior without poisoning unrelated kernels or backends. +- Confirm malformed DSL and genuine interpreter errors still surface normally. + +### 3. Python integration and reference examples + +Use existing suites such as `tests/ndarray/test_jit.py`, +`tests/ndarray/test_jit_dsl_dispatch.py`, and +`tests/ndarray/test_dsl_kernels.py`, with focused new cases where necessary. + +- Check `arange`, `linspace`, a plain expression, and a control-flow DSL kernel + against expected results under successful JIT and forced backend failure. +- Cover omitted settings, explicit `jit=True`/`False`, and explicit TCC/CC backend + selection. Verify documented environment precedence on each applicable API. +- Use subprocesses for loader/allocator failures so unintended process termination + is detected. Disable incidental Python bytecode writes when checking JIT files. +- Assert actual TCC JIT success in the positive case, not just successful output + that could have come from the interpreter. Test fallback explicitly as well. +- Run the documented API examples and relevant ndarray doctests. Check the + generated `arange`/`linspace` reference displays both JIT controls. + +### 4. Real platform and policy coverage + +| Environment | Required observation | +| --- | --- | +| RHEL 10, SELinux enforcing, policy permitting executable memfd mappings | Installed wheel executes a real TCC kernel successfully | +| SELinux enforcing policy denying the requested executable mapping | Correct interpreter result, quiet default output, explanatory opt-in trace | +| Linux with a `noexec` temporary mount | TCC succeeds independently of that mount when memfd execution is allowed | +| Linux with unusable temporary/cache storage | TCC has no filesystem-cache dependency | +| Linux with memfd unavailable or denied by the environment | Quiet, recoverable interpreter fallback | +| Linux x86_64 and ARM64 | Correct relocation, execution, cache synchronization, and cleanup | +| macOS and supported Windows native targets | In-memory TCC execution and quiet fallback still work | +| Existing interpreter-only targets and WebAssembly | Existing backend policies and validation behavior remain supported | + +Run SELinux-specific cases on an enforcing Linux host or VM. An ordinary container +on a host without enforcing SELinux cannot establish this behavior. Use targeted +syscall/mapping inspection to verify memfd use and absence of JIT artifact writes; +distinguish normal shared-library reads from generated-file activity. + +### 5. Performance checks + +- Measure small fresh kernels and repeated construction, plus a representative + large array, to check memfd setup and code publication overhead. +- Compare TCC before/after and CC cache-cold/cache-warm runs separately. +- Include more than a scalar ramp: control flow, indexed kernels, and math calls + can have different compile and execution costs. +- Measure repeated denied attempts to ensure negative caching bounds overhead. +- Report results without timing-based correctness assertions or changing the + default backend based on this narrow benchmark set. + +## Implementation order and completion checks + +1. Implement and test minicc's Linux allocator and build configuration. +2. Implement diagnostic capture and filesystem-independent TCC dispatch in miniexpr. +3. Verify CC fallback behavior and fix the generated allocation declaration. +4. Update dependency pins and exercise the integrated Python paths. +5. Complete reference documentation, docstrings, examples, and release notes. +6. Validate the installed Linux wheel under enforcing SELinux and run platform + regressions before considering the issue resolved. + +Use the `blosc2` conda environment for all Python, test, and build/install commands. +Run relevant native C tests in each dependency, focused Python tests first, and +then the project's required default suite. Build docs with Sphinx and check links +and warnings. + +### Acceptance checklist + +- [ ] The default wheel needs neither a system compiler nor libselinux. +- [ ] Linux TCC uses memfd-backed RW/RX allocation with correct lifetime handling. +- [ ] TCC JIT creates no source, binary, metadata, or cache-directory artifacts. +- [ ] Invalid temporary/cache storage does not prevent a permitted TCC JIT. +- [ ] Policy-denied TCC and CC execution falls back quietly and computes correctly. +- [ ] Debug tracing explains why JIT was unavailable and identifies fallback. +- [ ] Real input errors retain their documented behavior. +- [ ] CC still supports explicit selection and persistent cache reuse. +- [ ] The reported generated `malloc` warning is resolved. +- [ ] `arange`, `linspace`, and computation reference pages expose the JIT options, + defaults, backend requirements, environment precedence, and fallback semantics. +- [ ] Linux policy tests and native platform regressions cover actual JIT success + as well as interpreter fallback. diff --git a/plans/safer-jit-review.md b/plans/safer-jit-review.md new file mode 100644 index 000000000..b6d286456 --- /dev/null +++ b/plans/safer-jit-review.md @@ -0,0 +1,190 @@ +# Safer JIT implementation review + +Status: implementation completed and integrated on `safer-jit`; macOS validation +completed with the environment limitations recorded below. Linux policy/platform +verification remains a release gate. No dependency revisions were pushed. + +## Implemented + +- Linux minicc uses anonymous memfd-backed RX/RW aliases. It requests `MFD_EXEC`, + retries without that flag only for `EINVAL`, and reports allocation failures + without a filesystem fallback. Reservations, partial aliases and descriptors + are cleaned up. Section sizes are checked before narrowing; allocation sizes + before doubling. ARM/RISC-V publication synchronizes both aliases. +- TCC dispatch precedes all miniexpr filesystem-cache operations. Its states are + program-owned, not disk-cached. Failures keep the interpreter path; libtcc + diagnostics are captured on the owning program. API discovery and compile/delete + are serialized. Explicit libtcc path overrides do not substitute another library. +- Explicit CC compilation/loading and persistent cache reuse remain available. + Native fallback tracing names the interpreter. Generated allocation declarations + use the target size type and the code-generation cache version is bumped. +- Python dependency pins, subprocess regression tests, JIT reference documentation, + constructor/computation docstrings and release notes are updated. + +## Implementation work by repository + +### minicc: executable allocation and lifetime + +- `tcc_memfd.h` implements Linux anonymous backing storage via the syscall interface, + avoiding a dependency on the newer glibc `memfd_create` symbol. It closes the + descriptor after mapping and preserves the original failure's `errno` on rollback. +- `tccrun.c` selects the allocator for native Linux in both CMake and configure + builds, retains both aliases until state destruction, checks section/alignment + sizes and publishes generated instructions through both cache addresses. + Calling `tcc_relocate()` twice now returns an error instead of exiting the host. +- `tests/memfd_test.c` injects allocation failures without public runtime switches; + `tests/libtcc_api_test.c` exercises executable code and mutable globals/BSS. + `CMakeLists.txt` registers the tests and `README` explains the allocation model. + +### miniexpr: dispatch, diagnostics and generated C + +- `src/dsl_jit_runtime_host.c` dispatches TCC before cache-directory creation or + cache probes. `src/dsl_jit_runtime_nonhost.c` also uses the captured failure + reason and explicitly identifies interpreter fallback in traces. +- `src/dsl_jit_backend_libtcc.c` loads the error-callback API, stores diagnostics + on the owning program, checks extra-option failures and serializes compiler + discovery/compilation/destruction with POSIX or Windows locks. It honors + `ME_DSL_JIT_LIBTCC_PATH` as an exclusive override. +- `src/dsl_jit_cgen.c` replaces `malloc(unsigned long long)` with a target-size + declaration; `src/dsl_jit_runtime_internal.h` advances the cache version to 9. +- `tests/test_dsl_jit_libtcc.c` covers real compiler diagnostics and retained-state + lifetime; `tests/test_dsl_jit_codegen.c` checks the allocation declaration. +- CMake links thread support, pins the updated minicc revision and tracks allocator + sources as dependencies of the staged library. `README.md` documents fallback, + TCC's filesystem independence and CC's separate persistent-cache requirements. + +### Python-Blosc2: integration, regressions and API reference + +- `CMakeLists.txt` pins the updated miniexpr revision. Integrated builds use the + sibling sources through FetchContent overrides while the revisions remain local. +- `tests/ndarray/test_safer_jit.py` runs fresh-process checks using + `safer_jit_probe.py` and the test-only `safer_jit_libtcc.c` relocation-denial + double. The probes cover constructors, expressions, DSL control flow and + reductions under successful JIT and interpreter fallback. +- `doc/reference/jit.rst` is the canonical JIT options reference. Constructor, + lazy-array and DSL reference pages link to it; `src/blosc2/ndarray.py` and + `src/blosc2/lazyexpr.py` expose accurate defaults and best-effort semantics. +- Reduction guidance is updated in both `doc/conf.py` and the generated + `doc/reference/reduction_functions.rst`, so Sphinx does not overwrite it. + `RELEASE_NOTES.md` records the behavior changes, and + `plans/safer-jit-execution.md` records the implementation status. + +### Implementation revisions + +| Repository | Revision | Work | +| --- | --- | --- | +| minicc | `ffbb8ebf` | Anonymous memfd aliases and allocator/API tests | +| miniexpr | `19200c6` | In-memory TCC dispatch, diagnostic capture and minicc integration | +| miniexpr | `d704f89` | Target-size allocation declarations and cache-version update | +| Python-Blosc2 | `d9b5ca15` | Dependency integration and fallback regressions | +| Python-Blosc2 | `c10a0cd8` | Independent external-compiler diagnostic opt-in test | +| Python-Blosc2 | `8c92213c` | JIT reference, docstrings, release notes and review | +| Python-Blosc2 | `d756211b` | Preserve reduction JIT guidance through documentation generation | + +## Validation on macOS ARM64 + +All Python, test and build commands used the `blosc2` conda environment. + +- minicc CTest: **11 passed**, including mutable data/BSS, 100 execution/destruction + cycles, repeated relocation errors and allocation fault injection. +- The allocator harness covers create/resize/reservation/RX/RW failures, cleanup, + unsupported flags, unavailable syscall, denial, bounds and absence of RWX + requests. On macOS it uses test-only unlinked-file backing and strips executable + permission because shared executable file mappings are denied there. **This does + not validate Linux memfd or ARM64 alias execution.** +- miniexpr CTest: **36 passed**, excluding the unrelated Node smoke test. The + unfiltered run failed only because Homebrew Node cannot load `libllhttp.9.3.dylib`. + A separate miniexpr build with TCC JIT disabled succeeds. +- Integrated editable Python builds succeed using local FetchContent overrides + for miniexpr/minicc and the existing sibling SLEEF/C-Blosc2 sources. +- Focused Python JIT/DSL suites: **116 passed**, including the initial **18** new + regressions. A final run of the expanded new suite reports **19 passed**. + These check real TCC builds without a compiler or usable TMPDIR, no TCC cache + files, quiet relocation/compile/load fallback with correct results, opt-in + diagnostics (including the independent compiler-output opt-in), + disabling/precedence and fresh-process CC disk-cache reuse. +- Default Python suite: **10,613 passed, 36 skipped, 4 failed**. All four failures + are Node-dependent JavaScript tests with the same missing Homebrew library. + No environment-wide repair was attempted. +- Ruff passes. Sphinx HTML generation succeeds, constructor HTML exposes both JIT + kwargs, and the JIT Python examples execute successfully. Existing documentation + warnings remain (duplicate objects, autosummary stubs, themes and unrelated + references); the build is not warning-clean. The new cross-reference was fixed. + +## Findings addressed in review + +- Check ELF-sized fields before conversion to `unsigned`, not just the final size. +- Never apply `MAP_FIXED` after a failed reservation. +- Keep the writable alias alive for mutable data/BSS, not only code emission. +- Keep error-callback contexts alive until retained TCC states are destroyed. +- Dispatch TCC before any filesystem-cache creation or probing. +- Return recoverable failures instead of terminating the host or printing default + compiler diagnostics. +- Document `BLOSC_ME_JIT` at its actual `LazyExpr.compute()` scope; `0` is not a + disabling value. Use `ME_DSL_JIT=0` for miniexpr runtime JIT. +- Update the reduction-reference generator as well as its output; otherwise a + Sphinx build removes the newly added JIT guidance. + +## Remaining weak points and validation gaps + +1. **Linux remains a release gate.** No configured Linux VM was available locally. + Test the installed wheel on enforcing RHEL 10, Linux x86_64 and ARM64, allowed + and denied memfd mappings, and `noexec` temporary storage. Syscall tracing should + confirm anonymous storage and no generated files. Fault injection cannot + establish SELinux compatibility. +2. **Dual aliases are not immutable code.** Neither mapping is RWX, but physical + backing remains writable through one alias and executable through the other; + the RX mapping also covers the allocation's data region. This is the selected + allocation model, not a sandbox or strict physical-page W^X. Stricter policies + may deny it; interpreter fallback is intentional. +3. **Windows/WebAssembly were not runtime-tested.** Their allocation paths are + preserved. Windows gains callback capture/serialization. Homebrew Node was + repaired by the user; the JavaScript glue smoke test now passes locally. +4. **Concurrency coverage is limited.** The new lock protects libtcc discovery and + compile/delete, not all pre-existing miniexpr global positive/negative caches. + A comprehensive threaded/sanitizer audit remains separate work. +5. **Compiler-library discovery remains process-scoped.** Its initial failure is + sticky and loaded libraries are retained. Set path/environment controls before + Python starts. Negative-cache suppression may give a generic recent-failure + reason rather than repeat the original diagnostic. +6. **CC cache security is unchanged.** Ownership, races, symlinks and cache trust + were not hardened. Choose suitably protected storage; TCC bypasses that cache. +7. **No Linux performance comparison was possible.** Correctness checks do not + establish memfd overhead or throughput. Cold compilation, first library loading, + process startup and computation need separate timing. +8. **Dependency publication is not part of local validation.** Pins reference local + dependency revisions. Remote builds cannot fetch those revisions until they are + published; local builds used explicit source overrides. + +Issue #730 is not platform-verified until item 1 is completed. + +## Follow-up: Python configuration and native call-local options + +Implemented process defaults (`set_jit_options`, `get_jit_options`) and nested +`contextvars` overrides (`jit_options`) for JIT enablement/backend, accuracy, +tracing, compiler command/flags, exact CC cache directories and compiler output. +Evaluation takes a settings snapshot without modifying the environment. Explicit +per-call and LazyUDF settings retain precedence over Python defaults; existing +environment overrides retain precedence where supported. See the consolidated +parameter/precedence documentation in `doc/reference/jit.rst`. + +Miniexpr accepts borrowed call-local settings through +`me_compile_nd_jit_options`; compiling-thread settings restore on success and +failure. Compiler-affecting settings now enter process-cache keys. CC-generated +file paths are shell-quoted, including spaces, apostrophes and dollar signs. +TCC remains filesystem-independent and ignores CC-specific settings. + +Validation: the full Python suite passed with **10,635 passed, 36 skipped**. +After the final compiler-path parsing improvement, the 36 focused configuration +and fallback tests and all 38 miniexpr tests passed; Ruff checks passed. Sphinx +builds successfully, with unrelated existing warnings and no JIT-page warnings +after fixing heading underlines. The TCC-disabled native build also succeeds. +Linux enforcing SELinux and Windows/WebAssembly runtime validation remain open. + +Current dependency integration pins published miniexpr revision +`3f4db93a628aedc662b7d27a5f35c97ffaeb0424`, matching `CMakeLists.txt`. It includes +the native options API and compiler-command parser fix, plus the subsequent native +DSL syntax extensions and chained-comparison support. The configuration-specific +validation above originally used `a1c950522b6da2b8f812c8fd89f21b7ce5446c51` through +the sibling source override. The later integrated syntax validation passed with +**10,737 Python tests passed, 36 skipped**, and **42 native tests passed**. diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index 65936b555..75e61e82a 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -587,6 +587,7 @@ def _raise(exc): from .c2array import c2context, C2Array, C2NDSource, ChunkAlreadyWritten, URLPath from .dsl_kernel import DSLSyntaxError, DSLKernel, dsl_kernel, validate_dsl, validate_dsl_jit +from .jit_config import get_jit_options, jit_options, set_jit_options from .lazyexpr import ( LazyExpr, lazyudf, @@ -876,6 +877,10 @@ def _raise(exc): "to_utf8", # Grouped reductions "group_reduce", + # JIT configuration + "get_jit_options", + "set_jit_options", + "jit_options", # Classes "C2Array", "C2NDSource", diff --git a/src/blosc2/blosc2_ext.pyx b/src/blosc2/blosc2_ext.pyx index cb4652d63..64affa623 100644 --- a/src/blosc2/blosc2_ext.pyx +++ b/src/blosc2/blosc2_ext.pyx @@ -705,6 +705,18 @@ cdef extern from "miniexpr.h": int me_compile(const char *expression, const me_variable *variables, int var_count, me_dtype dtype, int *error, me_expr **out) + ctypedef struct me_jit_options: + const char *compiler + const char *cflags + const char *cache_dir + int trace + int compiler_output + + int me_compile_nd_jit_options(const char *expression, const me_variable *variables, + int var_count, me_dtype dtype, int ndims, const int64_t *shape, + const int32_t *chunkshape, const int32_t *blockshape, int jit_mode, + const me_jit_options *options, int *error, me_expr **out) + int me_compile_nd_jit(const char *expression, const me_variable *variables, int var_count, me_dtype dtype, int ndims, const int64_t *shape, const int32_t *chunkshape, @@ -1043,6 +1055,23 @@ cdef inline void _free_me_udata_tables(me_udata* udata, b2nd_array_t** inputs_, free(udata) +cdef int _me_compile_nd_configured(const char *expression, const me_variable *variables, + int n, me_dtype dtype, int ndims, const int64_t *shape, const int32_t *chunkshape, + const int32_t *blockshape, int jit_mode, int *error, me_expr **out_expr) except *: + options = blosc2.jit_config.execution_options() + cdef bytes compiler_bytes = options["compiler"].encode("utf-8") if options["compiler"] is not None else b"" + cdef bytes flags_bytes = options["cflags"].encode("utf-8") if options["cflags"] is not None else b"" + cdef bytes cache_bytes = options["cache_dir"].encode("utf-8") if options["cache_dir"] is not None else b"" + cdef me_jit_options native_options + native_options.compiler = compiler_bytes if compiler_bytes else NULL + native_options.cflags = flags_bytes if flags_bytes else NULL + native_options.cache_dir = cache_bytes if cache_bytes else NULL + native_options.trace = int(options["trace"]) + native_options.compiler_output = int(options["compiler_output"]) + return me_compile_nd_jit_options(expression, variables, n, dtype, ndims, shape, + chunkshape, blockshape, jit_mode, &native_options, error, out_expr) + + cdef inline str _me_compile_status_name(int rc): if rc == ME_COMPILE_SUCCESS: return "ME_COMPILE_SUCCESS" @@ -4396,9 +4425,9 @@ cdef class NDArray: cdef int64_t* shape = &self.array.shape[0] cdef int32_t* chunkshape = &self.array.chunkshape[0] cdef int32_t* blockshape = &self.array.blockshape[0] - cdef int rc = me_compile_nd_jit(expression_bytes, variables, n, me_dtype, ndims, - shape, chunkshape, blockshape, jit_mode, - &error, &out_expr) + cdef int rc = _me_compile_nd_configured(expression_bytes, variables, n, me_dtype, ndims, + shape, chunkshape, blockshape, jit_mode, + &error, &out_expr) cdef str me_error_msg = _me_compile_error_details(rc, error) if rc == ME_COMPILE_ERR_INVALID_ARG_TYPE: raise TypeError(f"miniexpr does not support operand or output dtype: {expression_display}; details: {me_error_msg}") @@ -4478,6 +4507,10 @@ cdef class NDArray: var.context = NULL var.itemsize = v.dtype.itemsize if v.dtype.num in (18, 19) else 0 + backend = blosc2.jit_config.execution_options()["jit_backend"] + if backend in ("tcc", "cc"): + from blosc2.lazyexpr import _apply_jit_backend_pragma + expression = _apply_jit_backend_pragma(expression, inputs, backend) cdef bytes expression_bytes = ( (expression).encode("utf-8") if isinstance(expression, str) else expression ) @@ -4487,7 +4520,7 @@ cdef class NDArray: cdef int64_t* shape = &self.array.shape[0] cdef int32_t* chunkshape = &self.array.chunkshape[0] cdef int32_t* blockshape = &self.array.blockshape[0] - cdef int rc = me_compile_nd_jit(expression_bytes, variables, n, me_output_dtype, ndims, + cdef int rc = _me_compile_nd_configured(expression_bytes, variables, n, me_output_dtype, ndims, shape, chunkshape, blockshape, ME_JIT_ON, &error, &out_expr) for i in range(n): diff --git a/src/blosc2/dsl_compare.py b/src/blosc2/dsl_compare.py new file mode 100644 index 000000000..ac6394abd --- /dev/null +++ b/src/blosc2/dsl_compare.py @@ -0,0 +1,137 @@ +"""Lower comparison chains for JavaScript emission (native DSL handles its own).""" + +import ast +import copy + + +def _has_chain(node): + return any(isinstance(child, ast.Compare) and len(child.ops) > 1 for child in ast.walk(node)) + + +class _ComparisonLowerer: + def __init__(self, func): + self.names = {node.id for node in ast.walk(func) if isinstance(node, ast.Name)} + self.names.update(node.arg for node in ast.walk(func) if isinstance(node, ast.arg)) + self.counter = 0 + + def _capture(self, value, statements): + while True: + name = f"b2_chain_{self.counter}" + self.counter += 1 + if name not in self.names: + break + self.names.add(name) + statements.append(ast.Assign(targets=[ast.Name(id=name, ctx=ast.Store())], value=value)) + return ast.Name(id=name, ctx=ast.Load()) + + @staticmethod + def _assign(target, value): + return ast.Assign(targets=[ast.Name(id=target.id, ctx=ast.Store())], value=value) + + def _operand(self, node, statements): + prelude, value = self._expr(node) + statements.extend(prelude) + return self._capture(value, statements) + + def _expr(self, node): + if not _has_chain(node): + return [], node + statements = [] + if isinstance(node, ast.Compare): + left = self._operand(node.left, statements) + right = self._operand(node.comparators[0], statements) + result = self._capture( + ast.Compare(left=left, ops=[node.ops[0]], comparators=[right]), statements + ) + for op, operand in zip(node.ops[1:], node.comparators[1:], strict=True): + guarded = [] + next_value = self._operand(operand, guarded) + guarded.append( + self._assign(result, ast.Compare(left=right, ops=[op], comparators=[next_value])) + ) + statements.append(ast.If(test=result, body=guarded, orelse=[])) + right = next_value + return statements, result + if isinstance(node, ast.BoolOp): + first = self._operand(node.values[0], statements) + result = self._capture( + ast.Call(func=ast.Name(id="bool", ctx=ast.Load()), args=[first], keywords=[]), statements + ) + for operand in node.values[1:]: + guarded = [] + value = self._operand(operand, guarded) + guarded.append( + self._assign( + result, ast.Call(func=ast.Name(id="bool", ctx=ast.Load()), args=[value], keywords=[]) + ) + ) + test = result if isinstance(node.op, ast.And) else ast.UnaryOp(op=ast.Not(), operand=result) + statements.append(ast.If(test=test, body=guarded, orelse=[])) + return statements, result + if isinstance(node, ast.BinOp): + node.left = self._operand(node.left, statements) + node.right = self._operand(node.right, statements) + elif isinstance(node, ast.UnaryOp): + node.operand = self._operand(node.operand, statements) + elif isinstance(node, ast.Call): + node.args = [self._operand(arg, statements) for arg in node.args] + else: + raise ValueError(f"Cannot lower comparison chain in {type(node).__name__}") + return statements, node + + def block(self, body): + result = [] + for node in body: + result.extend(self._stmt(node)) + return result + + def _stmt(self, node): + if isinstance(node, ast.If): + prelude, node.test = self._expr(node.test) + node.body = self.block(node.body) + node.orelse = self.block(node.orelse) + return [*prelude, node] + if isinstance(node, ast.While): + prelude, test = self._expr(node.test) + node.body = self.block(node.body) + node.orelse = self.block(node.orelse) + if prelude: + # Re-evaluate at the top of every iteration, including continue. + node.test = ast.Constant(value=1) + stop = ast.If(test=ast.UnaryOp(op=ast.Not(), operand=test), body=[ast.Break()], orelse=[]) + node.body = [*prelude, stop, *node.body] + return [node] + if isinstance(node, ast.For): + prelude = [] + if _has_chain(node.iter): + node.iter.args = [self._operand(arg, prelude) for arg in node.iter.args] + node.body = self.block(node.body) + node.orelse = self.block(node.orelse) + return [*prelude, node] + if isinstance(node, ast.Assign | ast.Return | ast.Expr | ast.AugAssign) and node.value is not None: + prelude, value = self._expr(node.value) + if isinstance(node, ast.AugAssign) and prelude: + # Capture the previous target value before evaluating the RHS. + before = [] + left = self._capture(ast.Name(id=node.target.id, ctx=ast.Load()), before) + assign = ast.Assign( + targets=[node.target], value=ast.BinOp(left=left, op=node.op, right=value) + ) + return [*before, *prelude, assign] + node.value = value + return [*prelude, node] + return [node] + + +def lower_chained_comparisons(func): + """Return a lowered function AST, or the original AST when no chains exist. + + Generated names avoid every identifier in the function. Operand evaluation + is left-to-right, once per chain, and subsequent links run only when all + preceding comparisons succeed. Existing DSL Boolean operations return bool. + """ + if not _has_chain(func): + return func + func = copy.deepcopy(func) + func.body = _ComparisonLowerer(func).block(func.body) + return ast.fix_missing_locations(func) diff --git a/src/blosc2/dsl_js.py b/src/blosc2/dsl_js.py index a5179ab29..f2993ca7b 100644 --- a/src/blosc2/dsl_js.py +++ b/src/blosc2/dsl_js.py @@ -122,6 +122,9 @@ def _get_source(obj) -> str: class _Transpiler: def transpile(self, func: ast.FunctionDef): + from .dsl_compare import lower_chained_comparisons + + func = lower_chained_comparisons(func) self.params = [a.arg for a in func.args.args] used_index = self._collect_index_symbols(func) hoist = self._hoist_names(func) @@ -178,6 +181,8 @@ def _stmt(self, node, ind): return f"{pad}break;\n" if isinstance(node, ast.Continue): return f"{pad}continue;\n" + if isinstance(node, ast.Pass): + return f"{pad};\n" raise _DSLToJSError(f"unsupported statement: {type(node).__name__}") def _augassign(self, node): diff --git a/src/blosc2/dsl_kernel.py b/src/blosc2/dsl_kernel.py index da7f6ce5a..b19f1965a 100644 --- a/src/blosc2/dsl_kernel.py +++ b/src/blosc2/dsl_kernel.py @@ -505,21 +505,11 @@ def validate(self, func_node: ast.FunctionDef): self._args(func_node) if not func_node.body: self._err(func_node, "DSL kernel must have a body") - self._one_per_line(func_node.body) - for stmt in func_node.body: - self._stmt(stmt) - - def _one_per_line(self, body: list[ast.stmt]): - # G1: miniexpr parses one statement per line; `;`-joined siblings share a lineno. - prev = None + body = func_node.body + if ast.get_docstring(func_node, clean=False) is not None: + body = body[1:] for stmt in body: - if prev is not None and stmt.lineno == prev: - self._err( - stmt, - "Only one statement per line is supported in DSL kernels; " - "split ';'-joined statements onto separate lines", - ) - prev = stmt.lineno + self._stmt(stmt) def _err(self, node: ast.AST, msg: str, *, line: int | None = None, col: int | None = None): if line is None: @@ -559,6 +549,8 @@ def _check_input_assign(self, target: ast.Name): ) def _stmt(self, node: ast.stmt): # noqa: C901 + if isinstance(node, ast.Pass): + return if isinstance(node, ast.Assign): if len(node.targets) != 1 or not isinstance(node.targets[0], ast.Name): self._err(node, "Only simple assignments are supported in DSL kernels") @@ -584,8 +576,6 @@ def _stmt(self, node: ast.stmt): # noqa: C901 self._expr(node.test) if not node.body: self._err(node, "Empty if blocks are not supported in DSL kernels") - self._one_per_line(node.body) - self._one_per_line(node.orelse) for stmt in node.body: self._stmt(stmt) for stmt in node.orelse: @@ -607,7 +597,6 @@ def _stmt(self, node: ast.stmt): # noqa: C901 self._expr(arg) if not node.body: self._err(node, "Empty for-loop bodies are not supported in DSL kernels") - self._one_per_line(node.body) for stmt in node.body: self._stmt(stmt) return @@ -617,7 +606,6 @@ def _stmt(self, node: ast.stmt): # noqa: C901 self._expr(node.test) if not node.body: self._err(node, "Empty while-loop bodies are not supported in DSL kernels") - self._one_per_line(node.body) for stmt in node.body: self._stmt(stmt) return @@ -648,11 +636,11 @@ def _expr(self, node: ast.AST): # noqa: C901 self._expr(value) return if isinstance(node, ast.Compare): - if len(node.ops) != 1 or len(node.comparators) != 1: - self._err(node, "Chained comparisons are not supported in DSL") - self._cmpop(node.ops[0]) + for op in node.ops: + self._cmpop(op) self._expr(node.left) - self._expr(node.comparators[0]) + for operand in node.comparators: + self._expr(operand) return if isinstance(node, ast.Call): self._call_name(node.func) @@ -1206,6 +1194,9 @@ def _args(self, args: ast.arguments): return names def _stmt(self, node: ast.stmt, indent: int): + if isinstance(node, ast.Pass): + self._emit("pass", indent) + return if isinstance(node, ast.Assign): if len(node.targets) != 1 or not isinstance(node.targets[0], ast.Name): raise ValueError("Only simple assignments are supported in DSL kernels") @@ -1345,6 +1336,8 @@ def _expr(self, node: ast.AST) -> str: # noqa: C901 return f"({left} {op} {right})" if isinstance(node, ast.BoolOp): op = "&" if isinstance(node.op, ast.And) else "|" + if any(isinstance(n, ast.Compare) and len(n.ops) > 1 for n in ast.walk(node)): + op = "and" if isinstance(node.op, ast.And) else "or" values = [self._expr(v) for v in node.values] expr = values[0] for val in values[1:]: @@ -1352,7 +1345,10 @@ def _expr(self, node: ast.AST) -> str: # noqa: C901 return expr if isinstance(node, ast.Compare): if len(node.ops) != 1 or len(node.comparators) != 1: - raise ValueError("Chained comparisons are not supported in DSL") + parts = [self._expr(node.left)] + for op, operand in zip(node.ops, node.comparators, strict=True): + parts.extend((self._cmpop(op), self._expr(operand))) + return f"({' '.join(parts)})" left = self._expr(node.left) right = self._expr(node.comparators[0]) op = self._cmpop(node.ops[0]) @@ -1428,6 +1424,8 @@ def _args(self, args: ast.arguments): return names def _stmt(self, node: ast.stmt) -> bool: # noqa: C901 + if isinstance(node, ast.Pass): + return True if isinstance(node, ast.Assign): if len(node.targets) != 1 or not isinstance(node.targets[0], ast.Name): return False diff --git a/src/blosc2/jit_config.py b/src/blosc2/jit_config.py new file mode 100644 index 000000000..b9b03ebf5 --- /dev/null +++ b/src/blosc2/jit_config.py @@ -0,0 +1,178 @@ +"""JIT defaults and task-local evaluation settings, without environment mutation.""" + +import contextlib +import contextvars +import functools +import os +import shlex +import threading + +import blosc2 + + +class _Unset: + def __repr__(self): + return "" + + +_UNSET = _Unset() +_BUILTIN = { + "jit": None, + "jit_backend": None, + "fp_accuracy": blosc2.FPAccuracy.DEFAULT, + "trace": False, + "compiler": None, + "cflags": None, + "cache_dir": None, + "compiler_output": False, +} +_defaults = _BUILTIN.copy() +_lock = threading.RLock() +_context = contextvars.ContextVar("blosc2_jit_options", default=None) +_execution = contextvars.ContextVar("blosc2_jit_execution", default=None) + + +def _validate_string(name, value): + compiler_path = name == "compiler" and isinstance(value, os.PathLike) + if name in ("compiler", "cache_dir"): + value = os.fspath(value) + if not isinstance(value, str): + raise TypeError(f"{name} must be a string or None") + if "\0" in value or (name != "cflags" and not value.strip()): + raise ValueError(f"{name} must not contain NUL or be empty") + if compiler_path: + return shlex.quote(value) + return os.path.abspath(value) if name == "cache_dir" else value + + +def _validate(values, *, check_platform=True): + result = {} + for name, value in values.items(): + if value is _UNSET: + continue + if name not in _BUILTIN: + raise TypeError(f"Unknown JIT option: {name}") + if value is None: + value = _BUILTIN[name] + if name in ("jit", "trace", "compiler_output"): + if value is not None and type(value) is not bool: + raise TypeError(f"{name} must be bool or None") + elif name == "jit_backend": + if value not in (None, "tcc", "cc", "js"): + raise ValueError("jit_backend must be None, 'tcc', 'cc', or 'js'") + if check_platform and value == "js" and not blosc2.IS_WASM: + raise ValueError("jit_backend='js' is only available under WebAssembly/Pyodide") + elif name == "fp_accuracy": + if not isinstance(value, blosc2.FPAccuracy): + raise TypeError("fp_accuracy must be a blosc2.FPAccuracy value or None") + elif value is not None: + value = _validate_string(name, value) + result[name] = value + return result + + +def set_jit_options( + *, + jit=_UNSET, + jit_backend=_UNSET, + fp_accuracy=_UNSET, + trace=_UNSET, + compiler=_UNSET, + cflags=_UNSET, + cache_dir=_UNSET, + compiler_output=_UNSET, +): + """Set process-wide JIT defaults and return the previous defaults as a dict. + + Omitted arguments leave settings unchanged; None resets a setting to its + built-in default. All settings are validated before any change is made. + Explicit evaluation settings and :func:`jit_options` contexts take precedence. + Native JIT remains best effort. See :ref:`JITOptions` for parameter details, + backend applicability, environment overrides and cache behavior. + """ + updates = _validate(locals()) + global _defaults + with _lock: + previous = _defaults.copy() + _defaults = {**_defaults, **updates} + return previous + + +def get_jit_options(): + """Return a copy of Python JIT defaults, including active context overrides. + + Does not include environment overrides or per-evaluation settings. + See :ref:`JITOptions`. + """ + with _lock: + defaults = _defaults.copy() + return {**defaults, **(_context.get() or {})} + + +@contextlib.contextmanager +def jit_options( + *, + jit=_UNSET, + jit_backend=_UNSET, + fp_accuracy=_UNSET, + trace=_UNSET, + compiler=_UNSET, + cflags=_UNSET, + cache_dir=_UNSET, + compiler_output=_UNSET, +): + """Override JIT defaults for this thread/async context temporarily. + + Accepts the same options as :func:`set_jit_options`. Omitted settings inherit; + None selects their built-in defaults. Nested contexts restore outer settings + on exit, including exceptions. Settings apply at evaluation time, not lazy + expression construction time. No environment variables are changed. + See :ref:`JITOptions` for parameters, examples and precedence. + """ + updates = _validate(locals()) + token = _context.set({**(_context.get() or {}), **updates}) + try: + yield get_jit_options() + finally: + _context.reset(token) + + +def execution_options(): + """Internal snapshot consumed synchronously by native compilation.""" + execution = _execution.get() + if execution is not None and execution[1] is _context.get(): + return execution[0].copy() + return get_jit_options() + + +def trace_enabled(): + env = os.environ.get("ME_DSL_TRACE") + return env != "0" if env else execution_options()["trace"] + + +def pop_execution_options(kwargs): + """Separate validated per-call execution options from storage kwargs. + + None inherits for evaluation APIs; it must still be removed from kwargs + before those kwargs are passed to a storage-only constructor. + """ + explicit = {name: kwargs.pop(name) for name in _BUILTIN if name in kwargs} + explicit = {name: value for name, value in explicit.items() if value is not None} + return _validate(explicit, check_platform=False) + + +def jit_execution(func): + """Resolve evaluation kwargs and strip non-storage execution options.""" + + @functools.wraps(func) + def wrapped(*args, **kwargs): + options = execution_options() + options.update(pop_execution_options(kwargs)) + token = _execution.set((options, _context.get())) + kwargs.update({name: options[name] for name in ("jit", "jit_backend", "fp_accuracy")}) + try: + return func(*args, **kwargs) + finally: + _execution.reset(token) + + return wrapped diff --git a/src/blosc2/lazyexpr.py b/src/blosc2/lazyexpr.py index fd48df9a2..732c1141e 100644 --- a/src/blosc2/lazyexpr.py +++ b/src/blosc2/lazyexpr.py @@ -561,7 +561,7 @@ def sort(self, order: str | list[str] | None = None) -> blosc2.LazyArray: def compute( self, item: slice | list[slice] | None = None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs: Any, ) -> blosc2.NDArray: """ @@ -589,10 +589,12 @@ def compute( WebAssembly prefer-js default, keeping it on miniexpr. - ``jit`` (bool | None): enable (``True``) or disable (``False``) JIT compilation - of the expression via miniexpr. When ``None`` (default), JIT is only used + of the expression via miniexpr. None inherits :ref:`JITOptions` defaults. + With built-in defaults, JIT is only used for DSL kernels; plain expressions are evaluated by the bytecode interpreter. - Setting ``jit=True`` forces auto-lift of plain expressions into JIT-compiled - kernels. + Setting ``jit=True`` requests auto-lift of plain expressions into DSL + kernels. Native JIT is best effort: compilation/allocation/loading + failures use the interpreter quietly. See :ref:`JITOptions`. - ``jit_backend`` (str | None): select the JIT compiler backend. Valid values are ``"tcc"`` (bundled Tiny C Compiler), ``"cc"`` (system C @@ -613,9 +615,10 @@ def compute( - ``BLOSC_ME_JIT`` environment variable: when set to ``"1"``, ``"true"``, ``"on"``, ``"tcc"``, or ``"cc"``, it forces ``jit=True`` and overrides - both the ``jit`` and ``jit_backend`` arguments — this lets you switch - JIT on or change backends from the command line without touching code. - Setting it to ``"tcc"`` or ``"cc"`` also selects that backend. + ``jit`` on paths calling ``LazyExpr.compute()``. Setting it to + ``"tcc"`` or ``"cc"`` additionally overrides ``jit_backend``. It is + not a universal constructor/LazyUDF override, and ``"0"`` does not + disable JIT. Use ``ME_DSL_JIT=0`` to disable miniexpr runtime JIT. - ``BLOSC_ME_JIT_TRACE`` environment variable: when set to ``"1"``, ``"true"``, or ``"on"``, prints a one-line diagnostic to stdout @@ -1577,34 +1580,36 @@ def _js_dtypes_ok(operands, kwargs) -> bool: The output dtype must be floating: integer/complex *output* goes to miniexpr (the bridge can't reproduce integer division/overflow/truncation semantics, and float64 can't hold - int64 exactly). Given a floating output, integer *inputs* are fine -- the bridge converts - every operand to float64, which is exactly what miniexpr does when promoting integer inputs - for a float result (so any values above 2**53 lose precision identically). Complex inputs - are rejected (the bridge is real-only).""" + int64 exactly). Smaller integer inputs are exactly representable in float64, + but 64-bit integer inputs must stay on miniexpr: comparisons can retain their + integer precision even with a float output. Complex inputs are rejected + (the bridge is real-only).""" dt = kwargs.get("dtype") if dt is None: # Inferred output: only safe when all operands are float (so the output is float too). - return all( - np.issubdtype(op.dtype, np.floating) - for op in operands.values() - if isinstance(op, blosc2.NDArray) - ) + return all(np.issubdtype(op.dtype, np.floating) for op in operands.values() if hasattr(op, "dtype")) if not np.issubdtype(np.dtype(dt), np.floating): return False return all( - np.issubdtype(op.dtype, np.floating) or np.issubdtype(op.dtype, np.integer) + np.issubdtype(op.dtype, np.floating) + or (np.issubdtype(op.dtype, np.integer) and op.dtype.itemsize < 8) for op in operands.values() - if isinstance(op, blosc2.NDArray) + if hasattr(op, "dtype") ) def _trace_js_backend(expression): """BLOSC_ME_JIT_TRACE counterpart for the JS bridge, which never reaches miniexpr's trace point in `fast_eval` (see there for the message format).""" - if os.environ.get("BLOSC_ME_JIT_TRACE", "").lower() in ("1", "true", "on"): + if blosc2.jit_config.trace_enabled() or os.environ.get("BLOSC_ME_JIT_TRACE", "").lower() in ( + "1", + "true", + "on", + ): source = getattr(expression, "dsl_source", None) or expression expr_short = str(source)[:120].replace("\n", " ") - print(f"[blosc2] engine=js expr={expr_short}", flush=True) + stream = sys.stderr if blosc2.jit_config.trace_enabled() else sys.stdout + print(f"[blosc2] engine=js expr={expr_short}", file=stream, flush=True) def _maybe_js_backend(expression, jit, jit_backend, reduce_args, operands, kwargs, shape=None): @@ -1702,6 +1707,7 @@ def _miniexpr_integer_atan2(expression, operands): ) +@blosc2.jit_config.jit_execution def fast_eval( # noqa: C901 expression: str | Callable[[tuple, np.ndarray, tuple[int]], None], operands: dict, @@ -1910,13 +1916,18 @@ def _miniexpr_eligible_operand(op): if is_dsl and not use_miniexpr: _raise_dsl_miniexpr_required(dsl_disable_reason) - if os.environ.get("BLOSC_ME_JIT_TRACE", "").lower() in ("1", "true", "on"): + if blosc2.jit_config.trace_enabled() or os.environ.get("BLOSC_ME_JIT_TRACE", "").lower() in ( + "1", + "true", + "on", + ): engine = ( "miniexpr" if use_miniexpr else ("ne_evaluate" if isinstance(expr_string, str) else "python-udf") ) jit_info = f"jit={jit}, backend={jit_backend}" if use_miniexpr else "" expr_short = str(expr_string)[:120].replace("\n", " ") - print(f"[blosc2] engine={engine} {jit_info} expr={expr_short}", flush=True) + stream = sys.stderr if blosc2.jit_config.trace_enabled() else sys.stdout + print(f"[blosc2] engine={engine} {jit_info} expr={expr_short}", file=stream, flush=True) if use_miniexpr: cparams = kwargs.pop("cparams", None) @@ -2674,6 +2685,7 @@ def step_handler(cslice, _slice): return out +@blosc2.jit_config.jit_execution def reduce_slices( # noqa: C901 expression: str | Callable[[tuple, np.ndarray, tuple[int]], None], operands: dict, @@ -3259,6 +3271,7 @@ def _eval_zero_input_dsl_if_needed( return True, full_res +@blosc2.jit_config.jit_execution def chunked_eval( # noqa: C901 expression: str | Callable[[tuple, np.ndarray, tuple[int]], None], operands: dict, item=(), **kwargs ): @@ -4027,7 +4040,7 @@ def sum( dtype=None, keepdims=False, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): if where is not None: @@ -4049,7 +4062,7 @@ def prod( dtype=None, keepdims=False, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): if where is not None: @@ -4065,14 +4078,18 @@ def prod( } return self.compute(_reduce_args=reduce_args, fp_accuracy=fp_accuracy, **kwargs) - def get_num_elements(self, axis, item): + def get_num_elements(self, axis, item, fp_accuracy=None, **execution_kwargs): if hasattr(self, "_where_args") and len(self._where_args) == 1: # We have a where condition, so we need to count the number of elements # fulfilling the condition orig_where_args = self._where_args self._where_args = {"_where_x": blosc2.ones(self.shape, dtype=np.int8)} - num_elements = self.sum(axis=axis, dtype=np.int64, item=item) - self._where_args = orig_where_args + try: + num_elements = self.sum( + axis=axis, dtype=np.int64, item=item, fp_accuracy=fp_accuracy, **execution_kwargs + ) + finally: + self._where_args = orig_where_args return num_elements # Compute the number of elements in the array shape = self.shape @@ -4091,9 +4108,10 @@ def mean( dtype=None, keepdims=False, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): + execution_kwargs = blosc2.jit_config.pop_execution_options(kwargs) where = self._normalize_where(where) expr = self if where is None else where.where(self, 0) item = kwargs.pop("item", ()) @@ -4103,11 +4121,14 @@ def mean( keepdims=keepdims, item=item, fp_accuracy=fp_accuracy, + **execution_kwargs, ) num_elements = ( - self.get_num_elements(axis, item) + self.get_num_elements(axis, item, fp_accuracy=fp_accuracy, **execution_kwargs) if where is None - else where.where(blosc2.ones(self.shape, dtype=np.int64), 0).sum(axis=axis, dtype=np.int64) + else where.where(blosc2.ones(self.shape, dtype=np.int64), 0).sum( + axis=axis, dtype=np.int64, fp_accuracy=fp_accuracy, **execution_kwargs + ) ) if np.isscalar(num_elements) and num_elements == 0: raise ValueError("mean of an empty array is not defined") @@ -4127,29 +4148,50 @@ def std( keepdims=False, ddof=0, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): + execution_kwargs = blosc2.jit_config.pop_execution_options(kwargs) where = self._normalize_where(where) item = kwargs.pop("item", ()) if item == (): # fast path mean_value = self.mean( - axis=axis, dtype=dtype, keepdims=True, where=where, fp_accuracy=fp_accuracy + axis=axis, + dtype=dtype, + keepdims=True, + where=where, + fp_accuracy=fp_accuracy, + **execution_kwargs, ) expr = (self - mean_value) ** 2 else: mean_value = self.mean( - axis=axis, dtype=dtype, keepdims=True, where=where, item=item, fp_accuracy=fp_accuracy + axis=axis, + dtype=dtype, + keepdims=True, + where=where, + item=item, + fp_accuracy=fp_accuracy, + **execution_kwargs, ) # TODO: Not optimal because we load the whole slice in memory. Would have to write # a bespoke std function that executed within slice_eval to avoid this probably. - expr = (self.slice(item) - mean_value) ** 2 - out = expr.mean(axis=axis, dtype=dtype, keepdims=keepdims, where=where, fp_accuracy=fp_accuracy) + expr = (self.compute(item, fp_accuracy=fp_accuracy, **execution_kwargs) - mean_value) ** 2 + out = expr.mean( + axis=axis, + dtype=dtype, + keepdims=keepdims, + where=where, + fp_accuracy=fp_accuracy, + **execution_kwargs, + ) if ddof != 0: num_elements = ( - self.get_num_elements(axis, item) + self.get_num_elements(axis, item, fp_accuracy=fp_accuracy, **execution_kwargs) if where is None - else where.where(blosc2.ones(self.shape, dtype=np.int64), 0).sum(axis=axis, dtype=np.int64) + else where.where(blosc2.ones(self.shape, dtype=np.int64), 0).sum( + axis=axis, dtype=np.int64, fp_accuracy=fp_accuracy, **execution_kwargs + ) ) out = np.sqrt(out * num_elements / (num_elements - ddof)) else: @@ -4169,29 +4211,50 @@ def var( keepdims=False, ddof=0, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): + execution_kwargs = blosc2.jit_config.pop_execution_options(kwargs) where = self._normalize_where(where) item = kwargs.pop("item", ()) if item == (): # fast path mean_value = self.mean( - axis=axis, dtype=dtype, keepdims=True, where=where, fp_accuracy=fp_accuracy + axis=axis, + dtype=dtype, + keepdims=True, + where=where, + fp_accuracy=fp_accuracy, + **execution_kwargs, ) expr = (self - mean_value) ** 2 else: mean_value = self.mean( - axis=axis, dtype=dtype, keepdims=True, where=where, item=item, fp_accuracy=fp_accuracy + axis=axis, + dtype=dtype, + keepdims=True, + where=where, + item=item, + fp_accuracy=fp_accuracy, + **execution_kwargs, ) # TODO: Not optimal because we load the whole slice in memory. Would have to write # a bespoke var function that executed within slice_eval to avoid this probably. - expr = (self.slice(item) - mean_value) ** 2 - out = expr.mean(axis=axis, dtype=dtype, keepdims=keepdims, where=where, fp_accuracy=fp_accuracy) + expr = (self.compute(item, fp_accuracy=fp_accuracy, **execution_kwargs) - mean_value) ** 2 + out = expr.mean( + axis=axis, + dtype=dtype, + keepdims=keepdims, + where=where, + fp_accuracy=fp_accuracy, + **execution_kwargs, + ) if ddof != 0: num_elements = ( - self.get_num_elements(axis, item) + self.get_num_elements(axis, item, fp_accuracy=fp_accuracy, **execution_kwargs) if where is None - else where.where(blosc2.ones(self.shape, dtype=np.int64), 0).sum(axis=axis, dtype=np.int64) + else where.where(blosc2.ones(self.shape, dtype=np.int64), 0).sum( + axis=axis, dtype=np.int64, fp_accuracy=fp_accuracy, **execution_kwargs + ) ) out = out * num_elements / (num_elements - ddof) out2 = kwargs.pop("out", None) @@ -4207,7 +4270,7 @@ def min( axis=None, keepdims=False, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): if where is not None: @@ -4228,7 +4291,7 @@ def max( axis=None, keepdims=False, where=None, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): if where is not None: @@ -4248,7 +4311,7 @@ def any( self, axis=None, keepdims=False, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): reduce_args = { @@ -4263,7 +4326,7 @@ def all( self, axis=None, keepdims=False, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): reduce_args = { @@ -4278,7 +4341,7 @@ def argmax( self, axis=None, keepdims=False, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): reduce_args = { @@ -4292,7 +4355,7 @@ def argmin( self, axis=None, keepdims=False, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): reduce_args = { @@ -4306,7 +4369,7 @@ def cumulative_sum( self, axis=None, include_initial: bool = False, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): reduce_args = { @@ -4322,7 +4385,7 @@ def cumulative_prod( self, axis=None, include_initial: bool = False, - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, **kwargs, ): reduce_args = { @@ -4650,7 +4713,7 @@ def explain(self) -> dict: def compute( self, item=(), - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, jit=None, jit_backend: str | None = None, **kwargs, @@ -5133,7 +5196,7 @@ def sort(self, order: str | list[str] | None = None) -> blosc2.LazyArray: def compute( self, item=(), - fp_accuracy: blosc2.FPAccuracy = blosc2.FPAccuracy.DEFAULT, + fp_accuracy: blosc2.FPAccuracy | None = None, jit=None, jit_backend=None, **kwargs, @@ -5178,6 +5241,8 @@ def compute( aux_kwargs["jit"] = jit if jit_backend is not None: aux_kwargs["jit_backend"] = jit_backend + if fp_accuracy is not None: + aux_kwargs["fp_accuracy"] = fp_accuracy urlpath = kwargs.get("urlpath") if urlpath is not None and urlpath == aux_kwargs.get( "urlpath", @@ -5403,14 +5468,18 @@ def myudf(inputs_tuple, output, offset): Whether to evaluate the function in chunks or not (blocks). jit: bool or None, optional JIT policy for miniexpr-backed execution: - ``None`` uses default behavior (currently, JIT is tried out), ``True`` prefers JIT, ``False`` disables JIT. + ``None`` inherits configured defaults (built-in DSL defaults try JIT), + ``True`` prefers JIT, ``False`` disables JIT. jit_backend: {"tcc", "cc", "js"} or None, optional JIT backend selection. ``None`` uses backend defaults (miniexpr "tcc"), except under WebAssembly where — unless ``jit=False`` — it *prefers* ``"js"`` for transpilable float DSL kernels and falls back to miniexpr otherwise (``jit=True`` prefers ``"js"`` - too, since it is JIT-compiled by the JS engine). ``"tcc"`` forces libtcc, ``"cc"`` - forces the C compiler backend, and ``"js"`` transpiles a :func:`blosc2.dsl_kernel` - to JavaScript (browser/Pyodide only; raises elsewhere). + too, since it is JIT-compiled by the JS engine). ``"tcc"`` selects bundled + libtcc without a disk cache, ``"cc"`` selects an installed C compiler with + persistent caching, and ``"js"`` transpiles a :func:`blosc2.dsl_kernel` + to JavaScript (browser/Pyodide only; raises elsewhere). Native JIT requests + remain best effort and use the interpreter if compilation or loading fails. + See :ref:`JITOptions` for requirements, diagnostics, and environment settings. kwargs: Any, optional Keyword arguments that are supported by the :func:`empty` constructor. These arguments will be used by the :meth:`LazyArray.__getitem__` and diff --git a/src/blosc2/ndarray.py b/src/blosc2/ndarray.py index 49d0fc6b3..98ab50b0a 100644 --- a/src/blosc2/ndarray.py +++ b/src/blosc2/ndarray.py @@ -6207,6 +6207,15 @@ def arange( Other Parameters ---------------- + jit: bool or None, optional + Execution policy for the DSL-backed ramp: None (default) inherits the + configured JIT policy (tries JIT with built-in defaults); True tries JIT, + and False disables it. Native compilation or allocation failure falls + back quietly to the interpreter. See :ref:`JITOptions`. + jit_backend: {"tcc", "cc", "js"} or None, optional + Select the runtime compiler. Native defaults use bundled TCC without + disk caching. CC needs an installed compiler and uses persistent caching; + JS is WebAssembly/Pyodide-only. See :ref:`JITOptions` for limitations. kwargs: dict, optional Keyword arguments that are supported by the :func:`empty` constructor. @@ -6264,6 +6273,7 @@ def ramp_arange(start, step): if is_inside_new_expr() or NUM == 0: # We already have the dtype and shape, so return immediately + blosc2.jit_config.pop_execution_options(kwargs) return blosc2.zeros(shape, dtype=dtype, **kwargs) # Windows and wasm32 does not support complex numbers in DSL @@ -6320,8 +6330,19 @@ def linspace( efficient, as it does not require an intermediate copy of the array. Default is True. **kwargs: Any - Keyword arguments accepted by the :func:`empty` constructor. + Storage arguments accepted by :func:`empty`, plus the execution options below. + Other Parameters + ---------------- + jit: bool or None, optional + Execution policy for the DSL-backed ramp: None (default) inherits the + configured JIT policy (tries JIT with built-in defaults); True tries JIT, + and False disables it. Native compilation or allocation failure falls + back quietly to the interpreter. See :ref:`JITOptions`. + jit_backend: {"tcc", "cc", "js"} or None, optional + Select the runtime compiler. Native defaults use bundled TCC without + disk caching. CC needs an installed compiler and uses persistent caching; + JS is WebAssembly/Pyodide-only. See :ref:`JITOptions` for limitations. Returns ------- @@ -6374,6 +6395,7 @@ def ramp_linspace(start, step): if is_inside_new_expr() or num == 0: # We already have the dtype and shape, so return immediately + blosc2.jit_config.pop_execution_options(kwargs) return blosc2.zeros(shape, dtype=dtype, **kwargs) # will return empty array for num == 0 # Windows and wasm32 does not support complex numbers in DSL diff --git a/src/blosc2/remote_store_cache.py b/src/blosc2/remote_store_cache.py index ef9a9f698..648e18e84 100644 --- a/src/blosc2/remote_store_cache.py +++ b/src/blosc2/remote_store_cache.py @@ -30,9 +30,9 @@ def lock_cache_file(file, *, blocking=False): if os.name == "nt": import msvcrt - if not file.tell(): - file.write(b"\0") - file.flush() + # Windows permits locking a byte beyond EOF. Do not initialize the byte: + # another owner may already have locked it, making even a write fail + # before LK_LOCK gets a chance to wait for ownership. file.seek(0) msvcrt.locking(file.fileno(), msvcrt.LK_LOCK if blocking else msvcrt.LK_NBLCK, 1) else: diff --git a/tests/ndarray/safer_jit_libtcc.c b/tests/ndarray/safer_jit_libtcc.c new file mode 100644 index 000000000..c2dbcd241 --- /dev/null +++ b/tests/ndarray/safer_jit_libtcc.c @@ -0,0 +1,26 @@ +/* Test-only libtcc double: inject recoverable relocation denial without public + failure switches or changing machine-wide security policy. */ +#include + +typedef struct { + void *opaque; + void (*error)(void *, const char *); +} TCCState; + +TCCState *tcc_new(void) { return calloc(1, sizeof(TCCState)); } +void tcc_delete(TCCState *s) { free(s); } +void tcc_set_error_func(TCCState *s, void *opaque, void (*error)(void *, const char *)) { + s->opaque = opaque; + s->error = error; +} +int tcc_set_output_type(TCCState *s, int type) { (void)s; (void)type; return 0; } +int tcc_compile_string(TCCState *s, const char *source) { (void)s; (void)source; return 0; } +int tcc_add_symbol(TCCState *s, const char *name, const void *value) { + (void)s; (void)name; (void)value; return 0; +} +int tcc_relocate(TCCState *s) { + if (s->error) + s->error(s->opaque, "tccrun: executable mapping failed: Permission denied"); + return -1; +} +void *tcc_get_symbol(TCCState *s, const char *name) { (void)s; (void)name; return NULL; } diff --git a/tests/ndarray/safer_jit_probe.py b/tests/ndarray/safer_jit_probe.py new file mode 100644 index 000000000..82c7e7254 --- /dev/null +++ b/tests/ndarray/safer_jit_probe.py @@ -0,0 +1,36 @@ +"""Fresh-process integration probe, invoked by test_safer_jit.py.""" + +import sys + +import numpy as np + +import blosc2 + + +@blosc2.dsl_kernel +def kernel(x): + acc = x + for i in range(3): + acc = acc + i + return acc * 2 + + +def main(): + backend, mode = sys.argv[1:] + jit = mode != "off" + kwargs = {"jit": jit, "jit_backend": backend} + a = blosc2.arange(0, 128, **kwargs) + np.testing.assert_array_equal(a[:], np.arange(128)) + a = blosc2.linspace(0, 10, 128, **kwargs) + np.testing.assert_allclose(a[:], np.linspace(0, 10, 128)) + x = np.arange(128, dtype=np.float64) + arr = blosc2.asarray(x) + result = (arr * 2 + 1).compute(**kwargs) + np.testing.assert_array_equal(result[:], x * 2 + 1) + result = blosc2.lazyudf(kernel, (arr,), dtype=np.float64, **kwargs).compute() + np.testing.assert_array_equal(result[:], (x + 3) * 2) + np.testing.assert_allclose((arr * 2 + 1).sum(**kwargs), (x * 2 + 1).sum()) + + +if __name__ == "__main__": + main() diff --git a/tests/ndarray/test_dsl_comparisons.py b/tests/ndarray/test_dsl_comparisons.py new file mode 100644 index 000000000..bc042657a --- /dev/null +++ b/tests/ndarray/test_dsl_comparisons.py @@ -0,0 +1,189 @@ +"""Comparison-chain semantics across lowering, interpreter and native backends.""" + +import ast + +import numpy as np +import pytest + +import blosc2 +from blosc2.dsl_compare import lower_chained_comparisons +from blosc2.dsl_kernel import kernel_from_source + + +@pytest.mark.parametrize("jit", [False, True]) +@pytest.mark.parametrize( + "expression", + [ + "0 <= x < 5", + "5 > x >= 0", + "x == 2 != 3", + "x != 2 == 2", + "-2 < x <= 4 != 3", + "(0 < x < 5) + (2 < x < 7)", + "not (0 < x < 5)", + "(x < 0) or (0 < x < 5)", + "(x > 0) and (0 < x < 5)", + "where(0 < x < 5, 1, 2)", + ], +) +def test_chained_comparisons(expression, jit): + source = f"def k(x):\n return {expression}\n" + kernel = kernel_from_source(source, "k") + assert "b2_chain_" not in kernel.dsl_source + assert expression in kernel.dsl_source + assert blosc2.validate_dsl(kernel)["valid"] + x = np.arange(-3, 8, dtype=np.float64) + namespace = {"where": np.where} + exec(source, namespace) + expected = np.array([namespace["k"](float(v)) for v in x], dtype=np.float64) + result = blosc2.lazyudf(kernel, (x,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], expected) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_chain_short_circuit_division(jit): + kernel = kernel_from_source("def k(x):\n return 0 < x < 12 / x\n", "k") + x = np.arange(-3, 8, dtype=np.int64) + expected = np.array([0 < int(v) < 12 / int(v) for v in x]) + result = blosc2.lazyudf(kernel, (x,), dtype=np.bool_, jit=jit) + np.testing.assert_array_equal(result[:], expected) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_chain_control_flow_and_continue(jit): + source = ( + "def k(x):\n" + " y = x\n" + " while 0 <= y < 3:\n" + " y += 1\n" + " continue\n" + " if 3 <= y < 5:\n" + " y += 10\n" + " elif -5 <= y < 0:\n" + " y -= 10\n" + " else:\n" + " y += 0 < y < 10\n" + " return y\n" + ) + kernel = kernel_from_source(source, "k") + namespace = {} + exec(source, namespace) + x = np.arange(-3, 8, dtype=np.float64) + expected = np.array([namespace["k"](float(v)) for v in x]) + result = blosc2.lazyudf(kernel, (x,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], expected) + + +@pytest.mark.parametrize( + ("values", "expected_calls"), + [ + ([0, 1, 2, 3], [0, 1, 2, 3]), + ([2, 1, 2, 3], [2, 1]), + ([0, 2, 1, 3], [0, 2, 1]), + ], +) +def test_lowering_evaluates_operands_once(values, expected_calls): + func = ast.parse("def k():\n return probe(0) < probe(1) < probe(2) < probe(3)\n").body[0] + lowered = lower_chained_comparisons(func) + calls = [] + + def probe(index): + calls.append(values[index]) + return values[index] + + namespace = {"probe": probe} + exec(compile(ast.Module(body=[lowered], type_ignores=[]), "", "exec"), namespace) + assert namespace["k"]() == (values[0] < values[1] < values[2] < values[3]) + assert calls == expected_calls + + +def test_temporary_names_do_not_collide(): + kernel = kernel_from_source( + "def k(b2_chain_0):\n b2_chain_1 = 7\n return 0 < b2_chain_0 < b2_chain_1\n", "k" + ) + x = np.arange(-1, 10, dtype=np.float64) + np.testing.assert_array_equal(blosc2.lazyudf(kernel, (x,), dtype=np.bool_)[:], (x > 0) & (x < 7)) + + +def test_chain_actual_native_jit(): + baseline = kernel_from_source("def k(x):\n return x + 1\n", "k") + available = blosc2.validate_dsl_jit(baseline, [np.float64], np.float64) + kernel = kernel_from_source("def k(x):\n return 0 < x < 5\n", "k") + status = blosc2.validate_dsl_jit(kernel, [np.float64], np.float64) + assert status["compiled"] + if available["jit"]: + assert status["jit"] + + +@pytest.mark.parametrize("jit", [False, True]) +def test_chain_boolean_output_preserves_float_operands(jit): + kernel = kernel_from_source("def k(x):\n return 0.25 < x < 0.75\n", "k") + x = np.array([np.nan, -np.inf, 0, 0.25, 0.5, 0.75, 1, np.inf]) + result = blosc2.lazyudf(kernel, (x,), dtype=np.bool_, jit=jit) + np.testing.assert_array_equal(result[:], (x > 0.25) & (x < 0.75)) + + +@pytest.mark.parametrize("jit", [False, True]) +@pytest.mark.parametrize("wasm_dispatch", [False, True]) +def test_chain_float_output_preserves_large_integer_operands(jit, wasm_dispatch, monkeypatch): + if wasm_dispatch: + monkeypatch.setattr(blosc2, "IS_WASM", True) + kernel = kernel_from_source("def k(x, lo, hi):\n return lo < x < hi\n", "k") + x = np.arange(6, dtype=np.int64) + 2**54 + lo = np.full(6, 2**54 + 1, dtype=np.int64) + hi = np.full(6, 2**54 + 4, dtype=np.int64) + result = blosc2.lazyudf(kernel, (x, lo, hi), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], (lo < x) & (x < hi)) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_chain_in_range_arguments(jit): + source = "def k(x):\n y = 0\n for i in range(int(0 < x < 5), 3):\n y += 1\n return y\n" + kernel = kernel_from_source(source, "k") + x = np.arange(-1, 7, dtype=np.float64) + expected = np.where((x > 0) & (x < 5), 2, 3) + np.testing.assert_array_equal(blosc2.lazyudf(kernel, (x,), dtype=np.float64, jit=jit)[:], expected) + + +def test_chain_while_iteration_cap_counts_body_only(monkeypatch): + monkeypatch.setenv("ME_DSL_WHILE_MAX_ITERS", "3") + kernel = kernel_from_source( + "def k(x):\n y = 0\n while 0 <= y < 3:\n y += 1\n return x + y\n", "k" + ) + x = np.arange(4, dtype=np.float64) + np.testing.assert_array_equal(blosc2.lazyudf(kernel, (x,), dtype=np.float64, jit=False)[:], x + 3) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_chain_string_operands(jit): + kernel = kernel_from_source("def k(x):\n return 'b' == x != 'a'\n", "k") + x = np.array(["a", "b", "c", "d", "e"]) + expected = (x == "b") & (x != "a") + np.testing.assert_array_equal(blosc2.lazyudf(kernel, (x,), dtype=np.bool_, jit=jit)[:], expected) + + +@pytest.mark.parametrize( + ("expression", "expected_calls"), + [ + ("probe(0) + (probe(1) < probe(2) < probe(3))", [0, 1, 2, 3]), + ("pair(probe(0), probe(1) < probe(2) < probe(3))", [0, 1, 2, 3]), + ("False and (probe(0) < probe(1) < probe(2))", []), + ("True or (probe(0) < probe(1) < probe(2))", []), + ("probe(0) < (probe(1) < probe(2) < probe(3)) < probe(4)", [0, 1, 2, 3, 4]), + ], +) +def test_lowering_nested_evaluation_order(expression, expected_calls): + func = ast.parse(f"def k():\n return {expression}\n").body[0] + calls = [] + + def probe(index): + calls.append(index) + return index + + namespace = {"probe": probe, "pair": lambda a, b: a + b} + exec( + compile(ast.Module(body=[lower_chained_comparisons(func)], type_ignores=[]), "", "exec"), + namespace, + ) + namespace["k"]() + assert calls == expected_calls diff --git a/tests/ndarray/test_dsl_js.py b/tests/ndarray/test_dsl_js.py index 2fe8e98c2..cca77ec49 100644 --- a/tests/ndarray/test_dsl_js.py +++ b/tests/ndarray/test_dsl_js.py @@ -79,6 +79,8 @@ def misc_dsl(x, y): def _run_node(module, pts, scalars): """Run the emitted JS over `pts` (list of input rows) and return the output list.""" + if blosc2.IS_WASM: + pytest.skip("emscripten cannot spawn the node subprocess") node = shutil.which("node") if not node: pytest.skip("node not found; skipping JS numeric-equivalence check") @@ -105,6 +107,40 @@ def _run_node(module, pts, scalars): return json.loads(res.stdout) +def test_pass_and_python_numeric_literals(): + def kernel(a): + pass + x = a + 1_000 + 0xFF + 0b1010 + 0o755 + if x > 0: + pass + else: + x += 1 + for _i in range(3): + pass + return x + + module = build_js_module(kernel) + points = [[-2000.0], [0.0], [3.0]] + np.testing.assert_array_equal(_run_node(module, points, []), [kernel(p[0]) for p in points]) + + +def test_chained_comparisons_and_short_circuit(): + source = ( + "def kernel(x):\n" + " y = x\n" + " while 0 <= y < 3:\n" + " y += 1\n" + " continue\n" + " return (0 < y < 10) and (0 < x < 10 // x)\n" + ) + namespace = {} + exec(source, namespace) + points = [[float(x)] for x in range(-3, 8)] + module = build_js_module(source) + expected = [bool(namespace["kernel"](p[0])) for p in points] + np.testing.assert_array_equal(_run_node(module, points, []), expected) + + def test_transpile_structure(): js_src, params = dsl_to_js(newton_dsl) assert params == ["a", "b", "max_iter", "relax"] @@ -168,6 +204,8 @@ def ramp(a): def _run_node_index(module, gshape, off, cshape, ncols=1): """Run an index-aware module over one block and return the (flat) output list.""" + if blosc2.IS_WASM: + pytest.skip("emscripten cannot spawn the node subprocess") node = shutil.which("node") if not node: pytest.skip("node not found; skipping JS numeric-equivalence check") @@ -259,7 +297,7 @@ def _idx(a): def test_prefer_js_selection(monkeypatch): monkeypatch.setattr(blosc2, "IS_WASM", True) af = blosc2.asarray(np.ones((4, 4), dtype=np.float64)) - ai = blosc2.asarray(np.ones((4, 4), dtype=np.int64)) + ai = blosc2.asarray(np.ones((4, 4), dtype=np.int32)) def sel(jit, jit_backend, operands, kwargs, reduce_args=None): return lx._maybe_js_backend(_add, jit, jit_backend, reduce_args or {}, operands, kwargs) @@ -296,6 +334,20 @@ def sel(jit, jit_backend, operands, kwargs, reduce_args=None): assert not lx._is_dsl_kernel_expression(expr) +@pytest.mark.parametrize("dtype", [np.int64, np.uint64]) +@pytest.mark.parametrize("jit", [None, True]) +@pytest.mark.parametrize("array", [np.asarray, blosc2.asarray]) +def test_prefer_js_preserves_wide_integer_inputs(dtype, jit, array, monkeypatch): + monkeypatch.setattr(blosc2, "IS_WASM", True) + operand = array(np.arange(6, dtype=dtype) + 2**54) + expr, resolved_jit, backend = lx._maybe_js_backend( + _add, jit, None, {}, {"a": operand, "b": operand}, {"dtype": np.float64} + ) + assert expr is _add + assert resolved_jit is jit + assert backend is None + + def test_prefer_js_index_needs_shape(monkeypatch): monkeypatch.setattr(blosc2, "IS_WASM", True) af = blosc2.asarray(np.ones((4, 4), dtype=np.float64)) diff --git a/tests/ndarray/test_dsl_kernels.py b/tests/ndarray/test_dsl_kernels.py index 3149c7c2b..3d6097fe0 100644 --- a/tests/ndarray/test_dsl_kernels.py +++ b/tests/ndarray/test_dsl_kernels.py @@ -6,6 +6,7 @@ ####################################################################### +import ast import subprocess import sys import tempfile @@ -1062,13 +1063,175 @@ def test_dsl_save_dictstore_operands(tmp_path): # G3 (variable name colliding with miniexpr codegen identifier) --- -def test_kernel_semicolon_statements_rejected(): +@pytest.mark.parametrize("jit", [False, True]) +def test_kernel_semicolon_statements(jit): # Source built from a string so the formatter cannot rewrite the ';'-join away. - result = validate_dsl( - kernel_from_source("def k(a, b):\n x = a * a; y = b * b\n return x + y\n", "k") + kernel = kernel_from_source("def k(a, b):\n x = a * a; y = b * b\n return x + y\n", "k") + assert validate_dsl(kernel)["valid"] + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64, np.float64], np.float64)) + a = np.arange(24, dtype=np.float64) + b = a + 1 + result = blosc2.lazyudf(kernel, (a, b), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], a * a + b * b) + + +@pytest.mark.parametrize("jit", [False, True]) +@pytest.mark.parametrize( + "doc", + [ + "'Single line; # comment-like text'", + '"Double quoted"', + '"""Multiple lines.\nNo indentation required here; # text\n End."""', + "'''Multiple\n lines with \\' quotes.'''", + 'r"Raw \\n text"', + 'u"Unicode text"', + ], +) +def test_kernel_docstrings(doc, jit): + kernel = kernel_from_source(f"def k(a):\n {doc}\n x = a + 1; return x\n", "k") + assert kernel.__doc__ == kernel.func.__doc__ + assert validate_dsl(kernel)["valid"] + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64], np.float64)) + a = np.arange(24, dtype=np.float64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], a + 1) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_kernel_semicolons_in_nested_blocks(jit): + kernel = kernel_from_source( + "def k(a):\n" + " x = a; y = 0\n" + " for i in range(3):\n" + " y += 1; x += y\n" + " if x > 6:\n" + " x += 2; x *= 3; # trailing comment\n" + " else:\n" + " x -= 1; x *= 2\n" + " return x;\n", + "k", ) - assert not result["valid"] - assert "one statement per line" in result["error"].lower() + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64], np.float64)) + a = np.arange(24, dtype=np.float64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], np.where(a + 6 > 6, (a + 8) * 3, (a + 5) * 2)) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_kernel_parenthesized_continuation_with_comments(jit): + kernel = kernel_from_source( + "def k(a):\n" + " x = (a + # punctuation in comment: ); ' \"\n" + " 1)\n" + " for i in range(\n" + " 3 # ); ignored\n" + " ):\n" + " x += i\n" + " if (x >\n" + " 4 # ); ignored\n" + " ):\n" + " x *= 2; x += (\n" + " 1 # ignored\n" + " )\n" + " return x\n", + "k", + ) + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64], np.float64)) + a = np.arange(24, dtype=np.float64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], np.where(a + 4 > 4, (a + 4) * 2 + 1, a + 4)) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_kernel_pass(jit): + kernel = kernel_from_source( + "def k(a):\n" + " pass; x = a\n" + " if x > 0:\n" + " pass\n" + " else:\n" + " x -= 1; pass\n" + " pass\n" + " for i in range(0b11):\n" + " pass\n" + " while x < 0:\n" + " x += 1; pass\n" + " return x + 0x_FF\n", + "k", + ) + assert validate_dsl(kernel)["valid"] + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64], np.float64)) + a = np.arange(-3, 5, dtype=np.float64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], np.maximum(a, 0) + 255) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_kernel_pass_loop_retains_iteration_variable(jit): + kernel = kernel_from_source( + "def k(a):\n for i in range(0b11):\n pass\n return a + i\n", "k" + ) + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64], np.float64)) + a = np.arange(12, dtype=np.float64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.float64, jit=jit) + np.testing.assert_array_equal(result[:], a + 2) + + +@pytest.mark.parametrize("jit", [False, True]) +@pytest.mark.parametrize( + "literal", + [ + "1_000", + "0_0", + "0b1010", + "0B_10_10", + "0o755", + "0O_7_5_5", + "0xff", + "0X_FF", + "0xCA_FE", + "1_000.2_5", + ".1_25", + "1_0.", + "1e1_0", + "1_2.5e-0_2", + "0_1.0", + "-0b1010", + "-0x_FF", + ], +) +def test_kernel_python_numeric_literals(literal, jit): + kernel = kernel_from_source(f"def k(a):\n return a + {literal}\n", "k") + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.float64], np.float64)) + a = np.arange(12, dtype=np.float64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.float64, jit=jit) + np.testing.assert_allclose(result[:], a + ast.literal_eval(literal)) + + +@pytest.mark.parametrize("jit", [False, True]) +def test_kernel_numeric_literals_in_conditions_and_ranges(jit): + kernel = kernel_from_source( + "def k(a):\n" + " x_1 = a\n" + " for i in range(0b_1, 0x_4, 0o_1):\n" + " x_1 += 1_0\n" + " if x_1 > 3_2:\n" + " x_1 += 0b1_0\n" + " return x_1\n", + "k", + ) + if jit: + _expect_jit(blosc2.validate_dsl_jit(kernel, [np.int64], np.int64)) + a = np.arange(12, dtype=np.int64) + result = blosc2.lazyudf(kernel, (a,), dtype=np.int64, jit=jit) + np.testing.assert_array_equal(result[:], np.where(a + 30 > 32, a + 32, a + 30)) def test_dsl_kernel_reassigning_input_param_rejected(): diff --git a/tests/ndarray/test_jit_options.py b/tests/ndarray/test_jit_options.py new file mode 100644 index 000000000..dda096266 --- /dev/null +++ b/tests/ndarray/test_jit_options.py @@ -0,0 +1,387 @@ +"""Global and scoped JIT policies, including native compiler configuration.""" + +import asyncio +import os +import shutil +import sys +from concurrent.futures import ThreadPoolExecutor + +import numpy as np +import pytest + +import blosc2 + + +@pytest.fixture(autouse=True) +def restore_options(monkeypatch): + previous = blosc2.get_jit_options() + for name in ( + "CC", + "CFLAGS", + "ME_DSL_TRACE", + "ME_DSL_JIT", + "ME_DSL_JIT_COMPILER", + "ME_DSL_JIT_DEBUG_CC", + "ME_DSL_JIT_CACHE_DIR", + "BLOSC_ME_JIT", + "BLOSC_ME_JIT_TRACE", + ): + monkeypatch.delenv(name, raising=False) + yield + blosc2.set_jit_options(**previous) + + +def test_set_get_and_atomic_validation(): + original = blosc2.get_jit_options() + assert blosc2.set_jit_options(jit=True, jit_backend="cc") == original + result = blosc2.get_jit_options() + assert result["jit"] is True + assert result["jit_backend"] == "cc" + result["jit"] = False + assert blosc2.get_jit_options()["jit"] is True + with pytest.raises(TypeError): + blosc2.set_jit_options(jit=False, trace="yes") + assert blosc2.get_jit_options()["jit"] is True + blosc2.set_jit_options(jit=None, jit_backend=None) + assert blosc2.get_jit_options() == original + + +@pytest.mark.parametrize( + "kwargs", + [ + {"jit": 1}, + {"jit_backend": "invalid"}, + {"fp_accuracy": 1}, + {"compiler": ""}, + {"cflags": "\0"}, + {"cache_dir": ""}, + ], +) +def test_invalid_options(kwargs): + with pytest.raises((TypeError, ValueError)): + blosc2.set_jit_options(**kwargs) + with pytest.raises((TypeError, ValueError)): + with blosc2.jit_options(**kwargs): + pass + + +def test_nested_context_and_exception(): + blosc2.set_jit_options(jit=True, jit_backend="cc") + with blosc2.jit_options(jit=False): + assert blosc2.get_jit_options()["jit_backend"] == "cc" + + def fail(): + with blosc2.jit_options(jit=None, jit_backend="tcc"): + assert blosc2.get_jit_options()["jit"] is None + raise RuntimeError("restore") + + with pytest.raises(RuntimeError): + fail() + assert blosc2.get_jit_options()["jit"] is False + assert blosc2.get_jit_options()["jit"] is True + + +def test_async_context_isolation(): + async def worker(backend): + with blosc2.jit_options(jit_backend=backend): + await asyncio.sleep(0) + return blosc2.get_jit_options()["jit_backend"] + + async def run(): + return await asyncio.gather(worker("cc"), worker("tcc")) + + assert asyncio.run(run()) == ["cc", "tcc"] + + +@pytest.mark.skipif(blosc2.IS_WASM, reason="emscripten cannot start threads") +def test_thread_context_isolation(): + with blosc2.jit_options(jit_backend="cc"), ThreadPoolExecutor(max_workers=1) as pool: + assert pool.submit(blosc2.get_jit_options).result()["jit_backend"] is None + + +@pytest.mark.parametrize("method", ["mean", "var", "std"]) +@pytest.mark.parametrize("axis", [None, 0]) +@pytest.mark.parametrize("keepdims", [False, True]) +@pytest.mark.parametrize("masked", [False, True]) +def test_statistical_reductions_forward_execution_options( + method, axis, keepdims, masked, tmp_path, monkeypatch +): + data = np.arange(48, dtype=np.float64).reshape(8, 6) / 4 + expr = blosc2.asarray(data) + 1 + mask = (data % 3) != 0 if masked else None + options = { + "jit": False, + "jit_backend": "cc", + "trace": False, + "compiler": "/no/compiler", + "cflags": "-O1", + "cache_dir": str(tmp_path / "cache"), + "compiler_output": False, + } + observed = [] + original = blosc2.LazyExpr.sum + + def capture(self, *args, **kwargs): + observed.append(kwargs.copy()) + return original(self, *args, **kwargs) + + monkeypatch.setattr(blosc2.LazyExpr, "sum", capture) + statistic_kwargs = {"ddof": 1} if method != "mean" else {} + storage_kwargs = {"urlpath": str(tmp_path / "result.b2nd"), "mode": "w"} if axis == 0 else {} + result = getattr(expr, method)( + axis=axis, + keepdims=keepdims, + where=mask, + fp_accuracy=blosc2.FPAccuracy.MEDIUM, + **statistic_kwargs, + **options, + **storage_kwargs, + ) + expected = getattr(np, method)( + data + 1, axis=axis, keepdims=keepdims, where=True if mask is None else mask, **statistic_kwargs + ) + if axis == 0: + assert isinstance(result, blosc2.NDArray) + assert (tmp_path / "result.b2nd").exists() + result = result[:] + np.testing.assert_allclose(result, expected) + assert observed + for kwargs in observed: + assert {name: kwargs.get(name) for name in options} == options + assert kwargs["fp_accuracy"] == blosc2.FPAccuracy.MEDIUM + assert "urlpath" not in kwargs + assert "mode" not in kwargs + + +@pytest.mark.parametrize("method", ["mean", "var", "std"]) +def test_statistical_reductions_execution_options_with_out(method): + data = np.arange(24, dtype=np.float64).reshape(4, 6) + expr = blosc2.asarray(data) + 1 + out = blosc2.empty((6,), dtype=np.float64) + result = getattr(expr, method)(axis=0, out=out, jit=True, trace=False) + assert result is out + np.testing.assert_allclose(out[:], getattr(np, method)(data + 1, axis=0)) + + +@pytest.mark.parametrize("method", ["mean", "var", "std"]) +def test_statistical_reductions_reject_invalid_execution_options(method): + expr = blosc2.asarray(np.arange(12, dtype=np.float64)) + 1 + with pytest.raises(TypeError, match="jit must be bool"): + getattr(expr, method)(jit="invalid") + + +@pytest.mark.parametrize("method", ["mean", "var", "std"]) +@pytest.mark.parametrize("axis", [None, 0]) +def test_statistical_reductions_slice_execution_options(method, axis, monkeypatch): + data = np.arange(48, dtype=np.float64).reshape(8, 6) + expr = blosc2.asarray(data) + 1 + observed = [] + original = blosc2.LazyExpr.compute + + def capture(self, *args, **kwargs): + observed.append(kwargs.copy()) + return original(self, *args, **kwargs) + + monkeypatch.setattr(blosc2.LazyExpr, "compute", capture) + result = getattr(expr, method)( + axis=axis, item=slice(1, 5), jit=False, trace=False, fp_accuracy=blosc2.FPAccuracy.MEDIUM + ) + np.testing.assert_allclose(result, getattr(np, method)(data[1:5] + 1, axis=axis)) + assert observed + for kwargs in observed: + assert kwargs["jit"] is False + assert kwargs["trace"] is False + assert kwargs["fp_accuracy"] == blosc2.FPAccuracy.MEDIUM + + +@pytest.mark.parametrize("constructor", ["arange", "linspace"]) +@pytest.mark.parametrize("dtype", [np.float64, np.complex128]) +@pytest.mark.parametrize("shape", [None, (0, 2)]) +@pytest.mark.parametrize("jit", [False, True, None]) +def test_empty_ramps_accept_execution_options(constructor, dtype, shape, jit, tmp_path, capfd): + options = { + "jit": jit, + "jit_backend": "cc", + "fp_accuracy": blosc2.FPAccuracy.HIGH, + "trace": True, + "compiler": "/no/compiler", + "cflags": "-invalid-unused-option", + "cache_dir": tmp_path / "unused-cache", + "compiler_output": True, + } + urlpath = tmp_path / "empty.b2nd" + args = (0,) if constructor == "arange" else (0, 1, 0) + result = getattr(blosc2, constructor)( + *args, dtype=dtype, shape=shape, urlpath=urlpath, mode="w", **options + ) + assert result.shape == ((0,) if shape is None else shape) + assert result.dtype == dtype + assert result[:].size == 0 + assert urlpath.exists() + assert not (tmp_path / "unused-cache").exists() + assert capfd.readouterr() == ("", "") + + +@pytest.mark.parametrize("constructor", ["arange", "linspace"]) +def test_empty_ramps_accept_inherited_execution_options(constructor): + options = dict.fromkeys(blosc2.get_jit_options()) + args = (0,) if constructor == "arange" else (0, 1, 0) + with blosc2.jit_options(jit=True, trace=True): + result = getattr(blosc2, constructor)(*args, **options) + assert result.shape == (0,) + + +@pytest.mark.parametrize("constructor", ["arange", "linspace"]) +@pytest.mark.parametrize( + ("options", "error"), + [ + ({"jit": "invalid"}, TypeError), + ({"trace": "invalid"}, TypeError), + ({"jit_backend": "invalid"}, ValueError), + ({"fp_accuracy": 1}, TypeError), + ({"cflags": "\0"}, ValueError), + ], +) +def test_empty_ramps_validate_execution_options(constructor, options, error): + args = (0,) if constructor == "arange" else (0, 1, 0) + with pytest.raises(error): + getattr(blosc2, constructor)(*args, **options) + + +@pytest.mark.parametrize("constructor", ["arange", "linspace"]) +def test_ramp_metadata_shortcut_strips_execution_options(constructor, monkeypatch): + monkeypatch.setattr(sys.modules["blosc2.ndarray"], "is_inside_new_expr", lambda: True) + args = (3,) if constructor == "arange" else (0, 1, 3) + result = getattr(blosc2, constructor)(*args, jit=True, jit_backend="tcc", trace=True) + assert result.shape == (3,) + np.testing.assert_array_equal(result[:], 0) + + +@blosc2.dsl_kernel +def kernel(x): + return x * 2 + 1 + + +@pytest.mark.skipif(blosc2.IS_WASM, reason="Native miniexpr prefilter observation") +def test_fp_accuracy_precedence(monkeypatch): + observed = [] + original = blosc2.NDArray._set_pref_expr + + def capture(self, *args, **kwargs): + observed.append(kwargs.get("fp_accuracy", args[2] if len(args) > 2 else None)) + return original(self, *args, **kwargs) + + monkeypatch.setattr(blosc2.NDArray, "_set_pref_expr", capture) + arr = blosc2.asarray(np.arange(32, dtype=np.float64)) + blosc2.set_jit_options(fp_accuracy=blosc2.FPAccuracy.HIGH) + expr = arr + 1 + expr.compute() + assert observed[-1] == blosc2.FPAccuracy.HIGH + with blosc2.jit_options(fp_accuracy=blosc2.FPAccuracy.MEDIUM): + expr.compute(fp_accuracy=blosc2.FPAccuracy.HIGH) + assert observed[-1] == blosc2.FPAccuracy.HIGH + udf = blosc2.lazyudf(kernel, (arr,), dtype=arr.dtype) + udf.compute(fp_accuracy=blosc2.FPAccuracy.HIGH) + assert observed[-1] == blosc2.FPAccuracy.HIGH + + +@pytest.mark.skipif(blosc2.IS_WASM or os.name == "nt", reason="Native POSIX TCC execution") +def test_evaluation_time_and_udf_policy(capfd): + arr = blosc2.asarray(np.arange(32, dtype=np.float64)) + expr = arr + 1 + udf = blosc2.lazyudf(kernel, (arr,), dtype=arr.dtype, jit=False) + with blosc2.jit_options(jit=True, jit_backend="tcc", trace=True): + np.testing.assert_array_equal(expr.compute()[:], np.arange(32) + 1) + assert "jit runtime built:" in capfd.readouterr().err + np.testing.assert_array_equal(udf[:], np.arange(32) * 2 + 1) + assert "reason=jit_mode=off" in capfd.readouterr().err + assert blosc2.get_jit_options()["trace"] is False + + +@pytest.mark.skipif(blosc2.IS_WASM or os.name == "nt", reason="Native POSIX CC backend") +def test_probe_uses_configured_backend(tmp_path, capfd): + with blosc2.jit_options(jit_backend="cc", compiler="/no/compiler", cache_dir=tmp_path, trace=True): + status = blosc2.validate_dsl_jit(kernel, (np.float64,), np.float64) + assert status["compiled"] + assert not status["jit"] + assert "c compiler unavailable" in capfd.readouterr().err + + +@pytest.mark.skipif(blosc2.IS_WASM or os.name == "nt", reason="Native POSIX CC backend") +def test_native_cc_options_and_cache_identity(tmp_path, capfd): + if shutil.which("cc") is None: + pytest.skip("CC unavailable") + before = dict(os.environ) + cache = tmp_path / "direct cache's $data" + with blosc2.jit_options(jit_backend="cc", compiler="cc", cflags="-O1", cache_dir=cache, trace=True): + blosc2.linspace(0, 9, 64) + cold = capfd.readouterr() + assert cold.out == "" + assert "compiler=cc" in cold.err + assert "jit runtime built:" in cold.err + assert list(cache.glob("kernel_*.meta")) + assert not (cache / "miniexpr-jit").exists() + blosc2.linspace(0, 9, 64) + assert "jit runtime hit:" in capfd.readouterr().err + blosc2.linspace(0, 9, 64, trace=False) + assert capfd.readouterr() == ("", "") + assert len(list(cache.glob("kernel_*.meta"))) == 1 + blosc2.linspace(0, 9, 64, cflags="-O2") + assert "jit runtime built:" in capfd.readouterr().err + assert len(list(cache.glob("kernel_*.meta"))) == 2 + blosc2.linspace(0, 9, 64, compiler="/no/compiler") + assert "c compiler unavailable" in capfd.readouterr().err + second = tmp_path / "second-cache" + blosc2.linspace(0, 9, 64, cache_dir=second) + assert "jit runtime built:" in capfd.readouterr().err + assert list(second.glob("kernel_*.meta")) + assert dict(os.environ) == before + + +@pytest.mark.skipif(blosc2.IS_WASM or os.name == "nt", reason="Native POSIX TCC execution") +def test_environment_overrides_and_tcc_ignores_cc_settings(tmp_path, capfd, monkeypatch): + monkeypatch.setenv("ME_DSL_JIT_COMPILER", "tcc") + with blosc2.jit_options( + jit_backend="cc", compiler="/no/compiler", cache_dir=tmp_path / "cache", trace=True + ): + blosc2.arange(64) + assert "compiler=tcc" in capfd.readouterr().err + assert not list(tmp_path.iterdir()) + monkeypatch.setenv("ME_DSL_TRACE", "0") + with blosc2.jit_options(trace=True): + blosc2.arange(67) + assert capfd.readouterr() == ("", "") + + +@pytest.mark.skipif(blosc2.IS_WASM or os.name == "nt", reason="POSIX compiler-output injection") +def test_compiler_output(tmp_path, capfd): + compiler = tmp_path / "bad cc's executable" + compiler.write_text("#!/bin/sh\necho scoped-compiler-output >&2\nexit 1\n") + compiler.chmod(0o700) + with blosc2.jit_options( + jit_backend="cc", compiler=compiler, cache_dir=tmp_path / "cache", compiler_output=True + ): + blosc2.linspace(0, 3, 71) + captured = capfd.readouterr() + assert "scoped-compiler-output" in captured.err + assert "jit runtime fallback:" not in captured.err + + +@pytest.mark.skipif(blosc2.IS_WASM or os.name == "nt", reason="Native POSIX CC backend") +@pytest.mark.parametrize( + ("variable", "value", "reason"), + [ + ("CC", "/no/env/compiler", "c compiler unavailable"), + ("CFLAGS", "-invalid-blosc2-option", "compilation failed"), + ], +) +def test_cc_environment_precedence(tmp_path, capfd, monkeypatch, variable, value, reason): + monkeypatch.setenv(variable, value) + monkeypatch.setenv("ME_DSL_JIT_CACHE_DIR", str(tmp_path / "env-cache")) + with blosc2.jit_options( + jit_backend="cc", compiler="cc", cflags="-O1", cache_dir=tmp_path / "python-cache", trace=True + ): + blosc2.linspace(0, 11, 79) + assert reason in capfd.readouterr().err + assert (tmp_path / "env-cache").exists() + assert not (tmp_path / "python-cache").exists() diff --git a/tests/ndarray/test_safer_jit.py b/tests/ndarray/test_safer_jit.py new file mode 100644 index 000000000..1974985d2 --- /dev/null +++ b/tests/ndarray/test_safer_jit.py @@ -0,0 +1,184 @@ +"""Best-effort native JIT without filesystem requirements or unsolicited output.""" + +import os +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest + +import blosc2 +from blosc2.lazyexpr import _jit_from_env + +pytestmark = pytest.mark.skipif( + blosc2.IS_WASM or sys.platform == "win32", reason="POSIX native JIT subprocess tests" +) +HERE = Path(__file__).resolve().parent + + +def run_probe(tmp_path, backend="tcc", mode="on", trace=False, **overrides): + env = os.environ.copy() + for name in list(env): + if name.startswith(("ME_DSL_", "BLOSC_ME_JIT")) or name in ("CC", "CFLAGS"): + env.pop(name) + env.update(PYTHONDONTWRITEBYTECODE="1", TMPDIR=str(tmp_path), ME_DSL_TRACE="1" if trace else "0") + env.update(overrides) + return subprocess.run( + [sys.executable, str(HERE / "safer_jit_probe.py"), backend, mode], + env=env, + text=True, + capture_output=True, + timeout=60, + check=True, + ) + + +@pytest.fixture +def denied_libtcc(tmp_path): + cc = shutil.which("cc") + if cc is None: + pytest.skip("Test double needs a C compiler") + library = tmp_path / ("denied.dylib" if sys.platform == "darwin" else "denied.so") + subprocess.run( + [ + cc, + "-dynamiclib" if sys.platform == "darwin" else "-shared", + "-fPIC", + str(HERE / "safer_jit_libtcc.c"), + "-o", + str(library), + ], + check=True, + capture_output=True, + text=True, + ) + return library + + +def test_tcc_no_cache_directory(tmp_path): + result = run_probe(tmp_path, trace=True, CC="/no/compiler", PATH="") + assert "compiler=tcc" in result.stderr + assert "jit runtime built:" in result.stderr + assert "source=disk-cache" not in result.stderr + assert list(tmp_path.iterdir()) == [] + + +def test_tcc_invalid_tmpdir(tmp_path): + invalid = tmp_path / "not-a-directory" + invalid.write_text("not a cache") + result = run_probe(tmp_path, trace=True, TMPDIR=str(invalid), CC="/no/compiler", PATH="") + assert "jit runtime built:" in result.stderr + assert list(tmp_path.iterdir()) == [invalid] + + +@pytest.mark.parametrize("trace", [False, True]) +def test_tcc_denied_mapping_is_quiet(tmp_path, denied_libtcc, trace): + result = run_probe(tmp_path, trace=trace, ME_DSL_JIT_LIBTCC_PATH=str(denied_libtcc)) + assert result.stdout == "" + if trace: + assert "jit runtime fallback: interpreter" in result.stderr + assert "executable mapping failed: Permission denied" in result.stderr + assert "jit runtime built:" not in result.stderr + else: + assert result.stderr == "" + assert list(tmp_path.iterdir()) == [denied_libtcc] + + +@pytest.mark.parametrize("trace", [False, True]) +def test_tcc_compile_failure_is_quiet(tmp_path, trace): + result = run_probe(tmp_path, trace=trace, ME_DSL_JIT_TCC_OPTIONS="-invalid-safer-jit-option") + assert result.stdout == "" + if trace: + assert "jit runtime fallback: interpreter" in result.stderr + assert "invalid-safer-jit-option" in result.stderr + else: + assert result.stderr == "" + assert list(tmp_path.iterdir()) == [] + + +def test_tcc_explicit_missing_library(tmp_path): + result = run_probe(tmp_path, trace=True, ME_DSL_JIT_LIBTCC_PATH=str(tmp_path / "missing")) + assert "failed to load libtcc" in result.stderr + assert "jit runtime fallback: interpreter" in result.stderr + assert "jit runtime built:" not in result.stderr + + +@pytest.mark.parametrize("backend", ["tcc", "cc"]) +def test_jit_off_does_not_compile(tmp_path, backend): + result = run_probe(tmp_path, backend=backend, mode="off", trace=True) + assert "jit runtime built:" not in result.stderr + assert "reason=jit_mode=off" in result.stderr + assert list(tmp_path.iterdir()) == [] + + +def test_runtime_disable_wins_over_jit_true(tmp_path): + result = run_probe(tmp_path, trace=True, ME_DSL_JIT="0") + assert "jit runtime built:" not in result.stderr + assert "disabled by environment" in result.stderr + assert list(tmp_path.iterdir()) == [] + + +@pytest.mark.parametrize("trace", [False, True]) +def test_cc_missing_compiler_is_quiet(tmp_path, trace): + result = run_probe(tmp_path, backend="cc", trace=trace, CC="/no/compiler") + assert result.stdout == "" + if trace: + assert "c compiler unavailable" in result.stderr + assert "jit runtime fallback: interpreter" in result.stderr + else: + assert result.stderr == "" + + +@pytest.mark.parametrize("failure", ["compile", "load"]) +@pytest.mark.parametrize("trace", [False, True]) +def test_cc_failure_is_quiet(tmp_path, failure, trace): + compiler = tmp_path / "test-cc" + compiler.write_text( + "#!/bin/sh\n" + + ( + "echo test-compiler-error >&2\nexit 1\n" + if failure == "compile" + else 'while [ "$#" -gt 0 ]; do\n' + ' if [ "$1" = "-o" ]; then shift; echo invalid-library > "$1"; exit 0; fi\n' + " shift\ndone\nexit 1\n" + ) + ) + compiler.chmod(0o700) + result = run_probe(tmp_path, backend="cc", trace=trace, CC=str(compiler)) + assert result.stdout == "" + assert "test-compiler-error" not in result.stderr + if trace: + assert "jit runtime fallback: interpreter" in result.stderr + assert ("compilation failed" if failure == "compile" else "load failed") in result.stderr + else: + assert result.stderr == "" + + +def test_cc_reuses_disk_cache_in_fresh_process(tmp_path): + if shutil.which("cc") is None: + pytest.skip("System compiler unavailable") + cold = run_probe(tmp_path, backend="cc", trace=True) + warm = run_probe(tmp_path, backend="cc", trace=True) + assert "jit runtime built:" in cold.stderr + assert "source=disk-cache" in warm.stderr + assert "jit runtime built:" not in warm.stderr + assert list((tmp_path / "miniexpr-jit").glob("kernel_*.meta")) + + +def test_cc_compiler_output_is_opt_in(tmp_path): + compiler = tmp_path / "test-cc" + compiler.write_text("#!/bin/sh\necho test-compiler-diagnostic >&2\nexit 1\n") + compiler.chmod(0o700) + result = run_probe(tmp_path, backend="cc", CC=str(compiler), ME_DSL_JIT_DEBUG_CC="1") + assert "test-compiler-diagnostic" in result.stderr + assert "jit runtime fallback:" not in result.stderr # Runtime tracing is independently disabled. + + +def test_blosc_me_jit_environment_values(monkeypatch): + monkeypatch.setenv("BLOSC_ME_JIT", "cc") + assert _jit_from_env(False, "tcc") == (True, "cc") + monkeypatch.setenv("BLOSC_ME_JIT", "1") + assert _jit_from_env(False, "tcc") == (True, "tcc") + monkeypatch.setenv("BLOSC_ME_JIT", "0") + assert _jit_from_env(True, "tcc") == (True, "tcc") diff --git a/tests/test_cache_lock.py b/tests/test_cache_lock.py new file mode 100644 index 000000000..4b1215f4a --- /dev/null +++ b/tests/test_cache_lock.py @@ -0,0 +1,26 @@ +"""Windows cache lock acquisition must not write to another owner's locked byte.""" + +import sys +from types import SimpleNamespace + +import pytest + +import blosc2.remote_store_cache as cache + + +@pytest.mark.parametrize("blocking", [False, True]) +def test_windows_cache_lock_does_not_initialize_byte(tmp_path, monkeypatch, blocking): + calls = [] + msvcrt = SimpleNamespace( + LK_LOCK=1, + LK_NBLCK=2, + locking=lambda fd, mode, length: calls.append((fd, mode, length)), + ) + monkeypatch.setitem(sys.modules, "msvcrt", msvcrt) + monkeypatch.setattr(cache, "os", SimpleNamespace(name="nt")) + path = tmp_path / "owner.lock" + with path.open("a+b") as file: + cache.lock_cache_file(file, blocking=blocking) + assert file.tell() == 0 + assert calls == [(file.fileno(), msvcrt.LK_LOCK if blocking else msvcrt.LK_NBLCK, 1)] + assert path.read_bytes() == b""