Code generation
write_module writes a function's C source and header to disk. The rendering functions return
those artifacts in memory. See the code generation guide for an example
that compiles and calls the result.
Normal numerical calls to a Function compile and cache it automatically. Use
Function.compile() to prepare it before the first evaluation. The compilation and
toolchain interfaces below support direct control of that process.
write_module accepts a Function or a Solver. Other rendering functions and
workspace_size take a Function. Pass solve.function when exporting a solver.
Rendering
CModule
dataclass
One rendered function: the header and the .c to write, plus what a consumer needs to
compile and call them. body is the translation unit. program is the optimized Program IR
that produced it. workspace_size is the entry's w[] length, and backends lists the
solver plugins that the function calls. lang picks the header language ("c" or
"cpp"), and casadi adds the CasADi-compatible symbols.
header, source and link_flags are rendered on first access. The JIT compiles body
and asks for none of them. For a big sparse function the header alone is larger than the source.
Source code in src/scaly/codegen/aot.py
| @dataclass(frozen=True)
class CModule:
"""One rendered function: the header and the ``.c`` to write, plus what a consumer needs to
compile and call them. ``body`` is the translation unit. ``program`` is the optimized Program IR
that produced it. ``workspace_size`` is the entry's ``w[]`` length, and ``backends`` lists the
solver plugins that the function calls. ``lang`` picks the header language (``"c"`` or
``"cpp"``), and ``casadi`` adds the CasADi-compatible symbols.
``header``, ``source`` and ``link_flags`` are rendered on first access. The JIT compiles ``body``
and asks for none of them. For a big sparse function the header alone is larger than the source.
"""
fun: Function
header_name: str
source_name: str
body: str
program: ProgramNode
workspace_size: int
backends: tuple[str, ...]
lang: str = "c"
casadi: bool = False
recipe: BuildRecipe = BuildRecipe()
@cached_property
def header(self) -> str:
return self.recipe.comment(self.source_name) + _render_header(self.fun, self.backends, self.workspace_size, lang=self.lang, casadi=self.casadi)
@cached_property
def source(self) -> str:
"""The ``.c`` as written: ``body`` behind an include of the paired C header. A C++ header
cannot be included from C, so under ``lang="cpp"`` the kernel is ``body`` alone."""
if self.lang == "cpp":
return self.body
include = self.header_name.replace("\\", "\\\\").replace('"', '\\"')
recipe = self.recipe.comment(self.source_name)
return recipe + f'#include "{include}"\n\n' + self.body.removeprefix(recipe)
@cached_property
def link_flags(self) -> tuple[str, ...]:
"""Compiler and linker flags for ``backends``: include, lib, rpath and ``-l`` flags, plus libraries required by the math recipe. Resolved on demand because ``solvers.paths`` raises ``SolverLibraryError`` when a
backend's library or header is missing: rendering has to stay possible on a machine without the
vendored solver stack, and against a backend that has no library at all (the fake backends in
``tests/solvers/test_registry.py``)."""
return (*backend_compile_flags(self.backends), *self.recipe.link_flags)
|
source
cached
property
The .c as written: body behind an include of the paired C header. A C++ header
cannot be included from C, so under lang="cpp" the kernel is body alone.
link_flags
cached
property
link_flags: tuple[str, ...]
Compiler and linker flags for backends: include, lib, rpath and -l flags, plus libraries required by the math recipe. Resolved on demand because solvers.paths raises SolverLibraryError when a
backend's library or header is missing: rendering has to stay possible on a machine without the
vendored solver stack, and against a backend that has no library at all (the fake backends in
tests/solvers/test_registry.py).
render_c_module
render_c_module(fun: Function, *, header_name: str | None = None, source_name: str | None = None, lang: str = 'c', casadi: bool = False, lanes: LaneCount = 'auto', dialect: CDialect = 'gnu', vector_libm: VectorLibm = 'none', reciprocal: bool = False, cpu: CpuLevel = 'generic') -> CModule
Render fun into its header / .c pair from a single lowering. The kernel is always C.
lang picks the header a caller includes (f.h or f.hpp) and casadi adds the
CasADi 3.8 compatible symbols to both.
Source code in src/scaly/codegen/aot.py
| def render_c_module(
fun: Function,
*,
header_name: str | None = None,
source_name: str | None = None,
lang: str = "c",
casadi: bool = False,
lanes: LaneCount = "auto",
dialect: CDialect = "gnu",
vector_libm: VectorLibm = "none",
reciprocal: bool = False,
cpu: CpuLevel = "generic",
) -> CModule:
"""Render ``fun`` into its header / ``.c`` pair from a single lowering. The kernel is always C.
``lang`` picks the header a caller includes (``f.h`` or ``f.hpp``) and ``casadi`` adds the
CasADi 3.8 compatible symbols to both."""
_check_lang(lang)
ctx, body = _render_observed(
fun,
casadi=casadi,
recipe=BuildRecipe(cpu=cpu, lanes=lanes, dialect=dialect, vector_libm=vector_libm, reciprocal=reciprocal),
source_name=source_name,
)
symbol = c_ident(fun.name)
return CModule(
fun=fun,
header_name=f"{symbol}.{'hpp' if lang == 'cpp' else 'h'}" if header_name is None else header_name,
source_name=f"{symbol}.c" if source_name is None else source_name,
body=body,
program=ctx.prog,
workspace_size=entry_workspace(fun, ctx.workspace_size, casadi=casadi),
backends=ctx.backends,
lang=lang,
casadi=casadi,
recipe=ctx.recipe,
)
|
render_c_source
render_c_source(fun: Function, *, casadi: bool = False, lanes: LaneCount = 'auto', dialect: CDialect = 'gnu', vector_libm: VectorLibm = 'none', reciprocal: bool = False, cpu: CpuLevel = 'generic') -> str
Render a standalone pointer-ABI C implementation of fun and its callees. casadi adds
the CasADi 3.8 compatible symbols.
A LoweringError for a function that cannot be lowered propagates. There is no fallback.
Source code in src/scaly/codegen/aot.py
| def render_c_source(
fun: Function,
*,
casadi: bool = False,
lanes: LaneCount = "auto",
dialect: CDialect = "gnu",
vector_libm: VectorLibm = "none",
reciprocal: bool = False,
cpu: CpuLevel = "generic",
) -> str:
"""Render a standalone pointer-ABI C implementation of ``fun`` and its callees. ``casadi`` adds
the CasADi 3.8 compatible symbols.
A ``LoweringError`` for a function that cannot be lowered propagates. There is no fallback.
"""
recipe = BuildRecipe(cpu=cpu, lanes=lanes, dialect=dialect, vector_libm=vector_libm, reciprocal=reciprocal)
return _render_observed(fun, casadi=casadi, recipe=recipe)[1]
|
render_c_api_header(fun: Function, *, lang: str = 'c', casadi: bool = False, lanes: LaneCount = 'auto', dialect: CDialect = 'gnu', vector_libm: VectorLibm = 'none', reciprocal: bool = False, cpu: CpuLevel = 'generic') -> str
Render the public header for fun: the ABI declarations, SZ_* constants, the typed
buffers of the chosen lang, the
sparse-output tables, and with casadi the CasADi query prototypes.
Source code in src/scaly/codegen/aot.py
| def render_c_api_header(
fun: Function,
*,
lang: str = "c",
casadi: bool = False,
lanes: LaneCount = "auto",
dialect: CDialect = "gnu",
vector_libm: VectorLibm = "none",
reciprocal: bool = False,
cpu: CpuLevel = "generic",
) -> str:
"""Render the public header for ``fun``: the ABI declarations, ``SZ_*`` constants, the typed
buffers of the chosen ``lang``, the
sparse-output tables, and with ``casadi`` the CasADi query prototypes."""
_check_lang(lang)
if casadi:
check_casadi_layout(fun)
ctx = _lower(fun, recipe=BuildRecipe(cpu=cpu, lanes=lanes, dialect=dialect, vector_libm=vector_libm, reciprocal=reciprocal))
return ctx.recipe.comment(f"{c_ident(fun.name)}.c") + _render_header(
ctx.fun, ctx.backends, entry_workspace(fun, ctx.workspace_size, casadi=casadi), lang=lang, casadi=casadi
)
|
write_module
write_module(fun: Function | Solver, out_dir: Path, *, lang: str = 'c', casadi: bool = False, lanes: LaneCount = 'auto', dialect: CDialect = 'gnu', vector_libm: VectorLibm = 'none', reciprocal: bool = False, cpu: CpuLevel = 'generic') -> CModule
Write fun's header / .c into out_dir and return the module. A Solver renders its function.
Source code in src/scaly/codegen/aot.py
| def write_module(
fun: Function | Solver,
out_dir: Path,
*,
lang: str = "c",
casadi: bool = False,
lanes: LaneCount = "auto",
dialect: CDialect = "gnu",
vector_libm: VectorLibm = "none",
reciprocal: bool = False,
cpu: CpuLevel = "generic",
) -> CModule:
"""Write ``fun``'s header / ``.c`` into ``out_dir`` and return the module. A ``Solver`` renders its ``function``."""
if isinstance(fun, Solver):
fun = fun.function
module = render_c_module(fun, lang=lang, casadi=casadi, lanes=lanes, dialect=dialect, vector_libm=vector_libm, reciprocal=reciprocal, cpu=cpu)
out_dir.mkdir(parents=True, exist_ok=True)
(out_dir / module.header_name).write_text(module.header)
(out_dir / module.source_name).write_text(module.source)
return module
|
workspace_size
workspace_size(fun: Function, *, casadi: bool = False, lanes: LaneCount = 'auto', dialect: CDialect = 'gnu', vector_libm: VectorLibm = 'none', reciprocal: bool = False, cpu: CpuLevel = 'generic') -> int
Doubles of scratch fun needs in w[], the value its header's SZ_W quotes.
CModule.workspace_size is the same number without a second lowering, so prefer it when the
module is already in hand.
Source code in src/scaly/codegen/aot.py
| def workspace_size(
fun: Function,
*,
casadi: bool = False,
lanes: LaneCount = "auto",
dialect: CDialect = "gnu",
vector_libm: VectorLibm = "none",
reciprocal: bool = False,
cpu: CpuLevel = "generic",
) -> int:
"""Doubles of scratch ``fun`` needs in ``w[]``, the value its header's ``SZ_W`` quotes.
``CModule.workspace_size`` is the same number without a second lowering, so prefer it when the
module is already in hand."""
return entry_workspace(
fun,
_lower(fun, recipe=BuildRecipe(cpu=cpu, lanes=lanes, dialect=dialect, vector_libm=vector_libm, reciprocal=reciprocal)).workspace_size,
casadi=casadi,
)
|
Render options
These keyword options apply to render_c_module, render_c_source, render_c_api_header,
write_module, and workspace_size.
| Option |
Default |
Meaning |
cpu |
"generic" |
CPU baseline recorded in the recipe. Accepted values are generic, native, x86-64-v3, x86-64-v4, and apple-m4. |
lanes |
"auto" |
Target-selected lane width. An integer 1, 2, 4, or 8 requests a fixed width. 1 disables range widening. |
dialect |
"gnu" |
GNU vector extensions. "c" emits scalar operations inside lane loops using C99 syntax. |
vector_libm |
"none" |
Per-lane scalar math calls. "glibc" selects supported x86-64 libmvec functions and requires dialect="gnu". |
reciprocal |
False |
Permits invariant x / y to become x * (1 / y). This can change rounding, overflow, and underflow. |
The lang option selects the header language. It does not select the kernel dialect.
module.recipe.cpu_flags contains the target compiler flags. module.link_flags includes
solver dependencies and the selected math library. Generated headers and sources carry a recipe
comment with GCC and Clang build commands. The recipe's target flags must match the object build.
Lanes, dialects and math libraries
The code generation guide
explains these options, and vector lanes shows the
loops they produce. With lanes="auto", generated source selects SCALY_LANES from the
compiler's target macros, and -DSCALY_LANES=1, 2, 4, or 8 overrides that selection. A
fixed integer option emits a fixed width instead. The width is a request. A loop can use fewer
lanes, and a loop that cannot be widened stays scalar. Input and output arrays keep the same
layout at every width, and the header's workspace size does not depend on it.
dialect="gnu" uses GNU vector types and compiler builtins, as provided by GCC and Clang.
dialect="c" uses scalar lane loops and is intended for C99 compilers without those extensions.
Both give the same buffer layouts. Typed buffers use 16-byte alignment in C11 and C++, and
natural alignment in C99.
With vector_libm="none", transcendentals call scalar libm once per lane. The glibc option
emits guarded declarations for the supported sin, cos, tan, exp, log, pow, and tanh
vector symbols. Guards check x86-64, glibc, and the target feature required by the vector width.
tanh also requires glibc 2.35 or later. Incompatible builds fail at compile time. Cross builds
whose C library lacks libmvec use vector_libm="none". The glibc recipe links -lmvec.
Vector math is a separate numerical policy. The glibc 2.39 manual documents a maximum error of
four units in the last place for x86-64 libmvec functions. Results can differ from scalar libm.
See glibc's math accuracy reference.
Widening keeps the order of the additions in a sum. With the same compiler flags and scalar math,
a reduction gives the same bits as the scalar code. Reciprocal and vector-libm options have their
own rounding behavior. General algebraic simplification has the limits described under
arithmetic semantics.
CPU recipes and the command line
generic adds no CPU-specific flag. native uses -march=native on x86 and -mcpu=native on
AArch64. The x86-64 levels use -march=x86-64-v3 and -march=x86-64-v4. apple-m4 uses
-mcpu=apple-m4. native describes a build for the current host. Distributable binaries need a
CPU baseline supported by every destination machine.
The command-line equivalents are --cpu, --lanes, --dialect, --vector-libm, and
--reciprocal:
uv run scaly_codegen my_model:kernel -o generated --cpu generic --lanes auto --vector-libm none
uv run scaly_codegen my_model:kernel -o generated --cpu x86-64-v3 --lanes 4 --dialect gnu
scaly_toolchain reports the compiler and its detected native recipe. The just-in-time compiler
uses that native recipe with a fixed lane width. It enables libmvec on supported glibc x86-64
hosts after checking the host version. SCALY_VECTOR_LIBM=none forces scalar libm.
SCALY_VECTOR_LIBM=glibc explicitly requests libmvec. The source recipe and compile and link flags
participate in the JIT cache key. See environment variables.
Application binary interface
c_api_signature
c_api_signature(symbol: str = 'f') -> str
The universal ABI entry signature, spelled for a given symbol name.
Source code in src/scaly/codegen/abi.py
| def c_api_signature(symbol: str = "f") -> str:
"""The universal ABI entry signature, spelled for a given symbol name."""
return C_API_SIGNATURE.replace(" f(", f" {symbol}(")
|
render_cpp_header(fun: Function, backends: tuple[str, ...], sz_w: int, *, casadi: bool, sparsities: tuple[SparsityPattern | None, ...]) -> str
The .hpp for fun. The kernel symbols are declared extern "C" inside the function's
namespace, since a namespace and a function cannot share the global name. C linkage keeps the
symbol unmangled, so f::f is the same entry a C caller reaches as f.
Source code in src/scaly/codegen/cpp.py
| def render_cpp_header(fun: Function, backends: tuple[str, ...], sz_w: int, *, casadi: bool, sparsities: tuple[SparsityPattern | None, ...]) -> str:
"""The ``.hpp`` for ``fun``. The kernel symbols are declared ``extern "C"`` inside the function's
namespace, since a namespace and a function cannot share the global name. C linkage keeps the
symbol unmangled, so ``f::f`` is the same entry a C caller reaches as ``f``."""
symbol = c_ident(fun.name)
inputs, outputs = buffer_idents(fun)
params = [f"const {i}_t& {i}" for i in inputs] + [f"{o}_t& {o}" for o in outputs] + ["workspace_t& workspace"]
arg_init = ", ".join(f"{i}.ptr()" for i in inputs) or "nullptr"
res_init = ", ".join(f"{o}.ptr()" for o in outputs) or "nullptr"
lines = [
"#pragma once",
"",
"#include <array>",
"#include <cassert>",
"#include <cstddef>",
*(["#include <cstdint>", "", *stats_c_defs()] if backends else []),
"",
*abi_status_defines(guarded=True),
*(["", *casadi_defines()] if casadi else []),
"",
f"#define {symbol}_SZ_ARG {len(fun.inputs)}",
f"#define {symbol}_SZ_RES {len(fun.outputs)}",
f"#define {symbol}_SZ_IW 0",
f"#define {symbol}_SZ_W {sz_w}",
"",
BUFFER_TEMPLATE,
"",
f"namespace {symbol} {{",
f"// The pointer ABI for {fun.name}; the same C symbols the C header declares.",
'extern "C" ' + c_api_signature(symbol) + ";",
*(f'extern "C" int {s}_stats(scaly_solver_stats* out);' for s in solver_stats_symbols(fun)),
*(f'extern "C" {decl}' for decl in (casadi_declarations(symbol) if casadi else [])),
"",
*(_buffer_alias(i, e.shape) for i, e in zip(inputs, fun.inputs, strict=True)),
*(_buffer_alias(o, e.shape) for o, e in zip(outputs, fun.outputs, strict=True)),
f"using workspace_t = Buffer<double, {symbol}_SZ_W>;",
f"constexpr int sz_arg = {symbol}_SZ_ARG;",
f"constexpr int sz_res = {symbol}_SZ_RES;",
f"constexpr int sz_iw = {symbol}_SZ_IW;",
f"constexpr int sz_w = {symbol}_SZ_W;",
"",
f"inline int call({', '.join(params)}) {{",
f" const double* arg[sz_arg > 0 ? sz_arg : 1] = {{{arg_init}}};",
f" double* res[sz_res > 0 ? sz_res : 1] = {{{res_init}}};",
f" return {symbol}(arg, res, nullptr, sz_w ? workspace.ptr() : nullptr, 0);",
"}",
]
for name, sp in zip(fun.output_names, sparsities, strict=True):
if sp is not None:
lines += ["", *_sparse_namespace(name, sp)]
lines += [f"}} // namespace {symbol}"]
return "\n".join(lines) + "\n"
|
The CasADi layer
check_casadi_layout
check_casadi_layout(fun: Function) -> None
CasADi buffers are column-major and scaly's are row-major. Scalars, vectors and compact sparse
outputs agree. A dense matrix with both dimensions above one would need a transpose, so it is
rejected here rather than hand the caller a transposed matrix.
Source code in src/scaly/codegen/casadi.py
| def check_casadi_layout(fun: Function) -> None:
"""CasADi buffers are column-major and scaly's are row-major. Scalars, vectors and compact sparse
outputs agree. A dense matrix with both dimensions above one would need a transpose, so it is
rejected here rather than hand the caller a transposed matrix."""
for kind, names, exprs, sparsities in (
("input", fun.input_names, fun.inputs, (None,) * len(fun.inputs)),
("output", fun.output_names, fun.outputs, fun.output_sparsities),
):
for name, expr, sp in zip(names, exprs, sparsities, strict=True):
if sp is None and (len(expr.shape) > 2 or (len(expr.shape) == 2 and min(expr.shape) > 1)):
raise ValueError(
f"casadi=True: {kind} {name!r} of {fun.name!r} has dense matrix shape {expr.shape}; CasADi is column-major, so only scalars, vectors and compact sparse outputs are supported"
)
|
casadi_sparsity
casadi_sparsity(shape: tuple[int, ...], sp: SparsityPattern | None) -> tuple[int, ...]
CasADi's compressed encoding: {nrow, ncol, 1} for dense, {nrow, ncol, colind..., row...}
for sparse. check_casadi_layout has already ruled out dense matrices.
Source code in src/scaly/codegen/casadi.py
| def casadi_sparsity(shape: tuple[int, ...], sp: SparsityPattern | None) -> tuple[int, ...]:
"""CasADi's compressed encoding: ``{nrow, ncol, 1}`` for dense, ``{nrow, ncol, colind..., row...}``
for sparse. ``check_casadi_layout`` has already ruled out dense matrices."""
if sp is None:
return (*_dense_dims(shape), 1)
col_ptr, row_ind, _ = sp.to_csc()
return (*sp.shape, *col_ptr, *row_ind)
|
casadi_scratch
casadi_scratch(fun: Function) -> int
Doubles added to SZ_W for the compressed-column gather: nnz per compact sparse output
whose native order is not already compressed-column (the QP path's patterns are, so they add 0).
Source code in src/scaly/codegen/casadi.py
| def casadi_scratch(fun: Function) -> int:
"""Doubles added to ``SZ_W`` for the compressed-column gather: ``nnz`` per compact sparse output
whose native order is not already compressed-column (the QP path's patterns are, so they add 0)."""
return sum(sp.nnz for sp in fun.output_sparsities if _needs_gather(sp) and sp is not None)
|
Compiling and caching
CompiledFunction
Handle around a JIT-compiled Function.
Holds the ctypes.CDLL for the cached shared object, the resolved entry point with
argtypes/restype set up for the pointer ABI, and the workspace size the rendered module
reported. That is the header's SZ_W, since the library exports no size query of its own.
Source code in src/scaly/codegen/jit.py
| class CompiledFunction:
"""Handle around a JIT-compiled `Function`.
Holds the ``ctypes.CDLL`` for the cached shared object, the resolved entry point with
``argtypes``/``restype`` set up for the pointer ABI, and the workspace size the rendered module
reported. That is the header's ``SZ_W``, since the library exports no size query of its own.
"""
__slots__ = (
"_fun",
"_artifact",
"_lib",
"_symbol",
"_entry",
"_stats_entries",
"_sz_w",
"_input_sizes",
"_input_shapes",
"_input_names",
"_output_shapes",
"_output_sizes",
"_n_args",
"_n_res",
)
def __init__(self, fun: Function):
self._fun = fun
self._artifact = _build_artifact(fun)
self._lib = load_library(self._artifact.lib_path, isolated=self._artifact.solver_library)
symbol = c_ident(fun.name)
self._symbol = symbol
entry = getattr(self._lib, symbol)
entry.argtypes = [
ctypes.POINTER(_C_DOUBLE_P),
ctypes.POINTER(_C_DOUBLE_P),
_C_INT_P,
_C_DOUBLE_P,
ctypes.c_int,
]
entry.restype = ctypes.c_int
self._entry = entry
self._stats_entries: dict[str, Any] = {}
for stats_symbol in solver_stats_symbols(fun):
try:
stats_entry = getattr(self._lib, f"{stats_symbol}_stats")
except AttributeError as exc:
raise JitError(f"compiled artifact is missing solver stats symbol {stats_symbol}_stats") from exc
stats_entry.argtypes = [ctypes.POINTER(CSolverStats)]
stats_entry.restype = ctypes.c_int
self._stats_entries[stats_symbol] = stats_entry
self._sz_w = self._artifact.workspace_size
self._input_names = tuple(fun.input_names)
self._input_shapes = tuple(e.shape for e in fun.inputs)
self._input_sizes = tuple(int(e.size) for e in fun.inputs)
self._output_shapes = tuple(e.shape for e in fun.outputs)
self._output_sizes = tuple(int(e.size) for e in fun.outputs)
self._n_args = len(fun.inputs)
self._n_res = len(fun.outputs)
@property
def lib_path(self) -> Path:
return self._artifact.lib_path
@property
def cache_key(self) -> str:
return self._artifact.key
def solver_stats(self, name: str | None = None) -> SolverStats:
if name is None:
if len(self._stats_entries) != 1:
raise JitError(f"solver name is required when an artifact has {len(self._stats_entries)} solver stats entries")
name = next(iter(self._stats_entries))
symbol = c_ident(name)
if symbol not in self._stats_entries:
raise JitError(f"no solver stats entry for {name!r}")
raw = CSolverStats()
status = self._stats_entries[symbol](ctypes.byref(raw))
if status != 0:
raise JitError(f"{symbol}_stats returned status {status}")
if raw.version == 0:
raise JitError(f"solver {name!r} has not run yet (stats version is 0)")
if raw.version != SCALY_SOLVER_STATS_VERSION:
raise JitError(f"solver stats ABI mismatch for {name!r}: artifact version {raw.version}, expected {SCALY_SOLVER_STATS_VERSION}")
return SolverStats.from_c(raw)
def run(self, args: list[np.ndarray]) -> list[np.ndarray]:
"""Dispatch the compiled entry point with positional NumPy inputs.
Inputs are converted to contiguous float64 buffers with shapes validated against the
declared input shapes. Outputs are freshly-allocated NumPy arrays reshaped to the
Function's declared output shapes (the compact buffer shape for sparse outputs).
Raises ``JitError`` if the C function returns a non-zero ABI status.
"""
if len(args) != self._n_args:
raise TypeError(f"expected {self._n_args} inputs, got {len(args)}")
arg_buffers: list[np.ndarray] = []
arg_array = (_C_DOUBLE_P * max(self._n_args, 1))()
for i, value in enumerate(args):
name = self._input_names[i]
expected_shape = self._input_shapes[i]
arr = np.asarray(value, dtype=np.float64)
if arr.shape != expected_shape:
raise ValueError(f"input {name!r} has shape {arr.shape}, expected {expected_shape}")
if not arr.flags["C_CONTIGUOUS"]:
arr = np.ascontiguousarray(arr)
arg_buffers.append(arr)
arg_array[i] = arr.ctypes.data_as(_C_DOUBLE_P)
outputs: list[np.ndarray] = []
res_array = (_C_DOUBLE_P * max(self._n_res, 1))()
for i in range(self._n_res):
out = np.empty(self._output_sizes[i], dtype=np.float64)
outputs.append(out)
res_array[i] = out.ctypes.data_as(_C_DOUBLE_P)
if self._sz_w:
w_buf = (ctypes.c_double * self._sz_w)()
w_ptr = ctypes.cast(w_buf, _C_DOUBLE_P)
else:
w_buf = None # noqa: F841 -- keep lifetime explicit even when unused
w_ptr = _C_DOUBLE_P()
status = self._entry(arg_array, res_array, _C_INT_P(), w_ptr, 0)
if status != 0:
raise JitError(f"{self._fun.name} returned ABI status {status}")
return [out.reshape(self._output_shapes[i]) for i, out in enumerate(outputs)]
|
run
run(args: list[ndarray]) -> list[np.ndarray]
Dispatch the compiled entry point with positional NumPy inputs.
Inputs are converted to contiguous float64 buffers with shapes validated against the
declared input shapes. Outputs are freshly-allocated NumPy arrays reshaped to the
Function's declared output shapes (the compact buffer shape for sparse outputs).
Raises JitError if the C function returns a non-zero ABI status.
Source code in src/scaly/codegen/jit.py
| def run(self, args: list[np.ndarray]) -> list[np.ndarray]:
"""Dispatch the compiled entry point with positional NumPy inputs.
Inputs are converted to contiguous float64 buffers with shapes validated against the
declared input shapes. Outputs are freshly-allocated NumPy arrays reshaped to the
Function's declared output shapes (the compact buffer shape for sparse outputs).
Raises ``JitError`` if the C function returns a non-zero ABI status.
"""
if len(args) != self._n_args:
raise TypeError(f"expected {self._n_args} inputs, got {len(args)}")
arg_buffers: list[np.ndarray] = []
arg_array = (_C_DOUBLE_P * max(self._n_args, 1))()
for i, value in enumerate(args):
name = self._input_names[i]
expected_shape = self._input_shapes[i]
arr = np.asarray(value, dtype=np.float64)
if arr.shape != expected_shape:
raise ValueError(f"input {name!r} has shape {arr.shape}, expected {expected_shape}")
if not arr.flags["C_CONTIGUOUS"]:
arr = np.ascontiguousarray(arr)
arg_buffers.append(arr)
arg_array[i] = arr.ctypes.data_as(_C_DOUBLE_P)
outputs: list[np.ndarray] = []
res_array = (_C_DOUBLE_P * max(self._n_res, 1))()
for i in range(self._n_res):
out = np.empty(self._output_sizes[i], dtype=np.float64)
outputs.append(out)
res_array[i] = out.ctypes.data_as(_C_DOUBLE_P)
if self._sz_w:
w_buf = (ctypes.c_double * self._sz_w)()
w_ptr = ctypes.cast(w_buf, _C_DOUBLE_P)
else:
w_buf = None # noqa: F841 -- keep lifetime explicit even when unused
w_ptr = _C_DOUBLE_P()
status = self._entry(arg_array, res_array, _C_INT_P(), w_ptr, 0)
if status != 0:
raise JitError(f"{self._fun.name} returned ABI status {status}")
return [out.reshape(self._output_shapes[i]) for i, out in enumerate(outputs)]
|
load_library
load_library(path: Path, *, isolated: bool) -> ctypes.CDLL
Load a shared library. isolated keeps Linux solver dependencies out of the host process linker namespace.
Source code in src/scaly/codegen/jit.py
| def load_library(path: Path, *, isolated: bool) -> ctypes.CDLL:
"""Load a shared library. ``isolated`` keeps Linux solver dependencies out of the host process linker namespace."""
global _SOLVER_NAMESPACE, _SOLVER_NAMESPACE_ANCHOR
if not isolated or sys.platform != "linux":
return ctypes.CDLL(str(path))
libc = ctypes.CDLL(None)
dlmopen = libc.dlmopen
dlmopen.argtypes = [ctypes.c_long, ctypes.c_char_p, ctypes.c_int]
dlmopen.restype = ctypes.c_void_p
with _SOLVER_NAMESPACE_LOCK:
handle = dlmopen(_LM_ID_NEWLM if _SOLVER_NAMESPACE is None else _SOLVER_NAMESPACE, os.fsencode(path), os.RTLD_NOW | os.RTLD_LOCAL)
if not handle:
raise OSError(f"dlmopen failed for {path}")
# Python 3.14 reopens a path even when CDLL receives a handle, so bind it directly.
lib = ctypes.CDLL(None)
lib._handle = handle
lib._name = str(path)
if _SOLVER_NAMESPACE is None:
dlinfo = libc.dlinfo
dlinfo.argtypes = [ctypes.c_void_p, ctypes.c_int, ctypes.c_void_p]
dlinfo.restype = ctypes.c_int
namespace = ctypes.c_long()
if dlinfo(handle, _RTLD_DI_LMID, ctypes.byref(namespace)) != 0:
raise OSError(f"dlinfo failed for {path}")
_SOLVER_NAMESPACE = namespace.value
_SOLVER_NAMESPACE_ANCHOR = lib
return lib
|
invalidate_cache
invalidate_cache(fun: Function) -> None
Drop both the in-memory artifact entry and the on-disk cache directory for fun.
Safe to call when nothing is cached yet. Codegen failures (NotImplementedError) are
swallowed, since there cannot be a corresponding cache entry to remove.
Source code in src/scaly/codegen/jit.py
| def invalidate_cache(fun: Function) -> None:
"""Drop both the in-memory artifact entry and the on-disk cache directory for ``fun``.
Safe to call when nothing is cached yet. Codegen failures (``NotImplementedError``) are
swallowed, since there cannot be a corresponding cache entry to remove.
"""
compiler = find_c_compiler()
if compiler is None:
return
try:
module = _render_native(fun, compiler.cc)
except NotImplementedError:
return
key = _compute_cache_key(module.body, fun_name=fun.name, compile_flags=(*compile_flags(module.recipe), *module.link_flags))
with _artifact_lock:
_artifact_cache.pop(key, None)
cache_dir = cache_root() / key
if cache_dir.exists():
shutil.rmtree(cache_dir, ignore_errors=True)
|
JitError
Bases: RuntimeError
Raised when a compiled function returns a non-zero ABI status code.
Source code in src/scaly/codegen/jit.py
| class JitError(RuntimeError):
"""Raised when a compiled function returns a non-zero ABI status code."""
|
JitUnavailable
Bases: RuntimeError
Raised when JIT compilation cannot proceed and the caller should fall back.
Source code in src/scaly/codegen/jit.py
| class JitUnavailable(RuntimeError):
"""Raised when JIT compilation cannot proceed and the caller should fall back."""
|
find_c_compiler() -> Compiler | None
Find the C compiler the JIT uses: SCALY_CC, else CC, else cc on PATH.
Returns the compiler path and which of those three supplied it, or None if nothing is found.
Source code in src/scaly/codegen/toolchain.py
| def find_c_compiler() -> Compiler | None:
"""Find the C compiler the JIT uses: ``SCALY_CC``, else ``CC``, else ``cc`` on ``PATH``.
Returns the compiler path and which of those three supplied it, or ``None`` if nothing is found.
"""
for key in ("SCALY_CC", "CC"):
override = os.environ.get(key)
if override:
found = shutil.which(override)
return Compiler(found, key) if found is not None else None
found = shutil.which("cc")
return Compiler(found, "PATH") if found is not None else None
|
Return the JIT cache directory: SCALY_CACHE_DIR, else $XDG_CACHE_HOME/scaly/jit, else ~/.cache/scaly/jit.
Source code in src/scaly/codegen/toolchain.py
| def cache_root() -> Path:
"""Return the JIT cache directory: ``SCALY_CACHE_DIR``, else ``$XDG_CACHE_HOME/scaly/jit``, else ``~/.cache/scaly/jit``."""
override = env_path("SCALY_CACHE_DIR")
if override is not None:
return override
xdg = env_path("XDG_CACHE_HOME")
base = xdg if xdg is not None else Path.home() / ".cache"
return base / "scaly" / "jit"
|