"""ctypes boundary; no subprocess or compilation in the request path.""" import ctypes import math import os from pathlib import Path class BendReducer: def __init__(self, library=None): path = Path(library or os.environ.get('JULIA_ROUTER_LIBRARY') or Path(__file__).parent / 'build/libjulia_router.so') try: self._lib = ctypes.CDLL(str(path.resolve())) except OSError as error: raise RuntimeError( f'Cannot load Bend runtime {path.resolve()}. From Julia-1 run ' 'python -m julia.router.build --native-cpu; or explicitly select ' 'transformer_backend="torch".') from error self._argmax = self._lib.julia_router_argmax self._argmax.argtypes = [ctypes.POINTER(ctypes.c_float), ctypes.c_uint32, ctypes.POINTER(ctypes.c_uint32)] self._argmax.restype = ctypes.c_int self._ptr = ctypes.POINTER(ctypes.c_float) self._softmax = self._lib.julia_router_softmax self._softmax.argtypes = [self._ptr, ctypes.c_uint32, self._ptr, ctypes.POINTER(ctypes.c_uint32)] self._softmax.restype = ctypes.c_int self._layernorm = self._lib.julia_router_layernorm self._layernorm.argtypes = [self._ptr, self._ptr, self._ptr, ctypes.c_uint32, ctypes.c_uint32, ctypes.c_float, self._ptr] self._layernorm.restype = ctypes.c_int def profile(self, *, enabled=False, reset=False): """Process-wide packed projection timings; opt in only for diagnostics.""" fn = self._lib.julia_router_profile fn.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.POINTER(ctypes.c_double)] fn.restype = None values = (ctypes.c_double * 4)() fn(enabled, reset, values) return dict(zip(('pack_ms', 'execute_ms', 'unpack_ms', 'calls'), values)) def argmax(self, scores): scores = tuple(scores) if not 1 <= len(scores) <= 1048576: raise ValueError('Expected 1–1048576 scores') if not all(math.isfinite(x) for x in scores): raise ValueError('Scores must be finite') values = (ctypes.c_float * len(scores))(*scores) result = ctypes.c_uint32() code = self._argmax(values, len(scores), ctypes.byref(result)) if code: raise ValueError(f'Bend rejected scores (status {code})') return result.value def softmax(self, scores): import numpy as np values = np.ascontiguousarray(scores, dtype=np.float32) if values.ndim != 1 or not 1 <= values.size <= 1048576: raise ValueError('Expected a nonempty vector of at most 1048576 scores') output = np.empty_like(values) index = ctypes.c_uint32() code = self._softmax(values.ctypes.data_as(self._ptr), values.size, output.ctypes.data_as(self._ptr), ctypes.byref(index)) if code: raise ValueError(f'Bend rejected scores (status {code})') return index.value, output def layernorm(self, values, gamma=None, beta=None, epsilon=1e-5): import numpy as np x = np.ascontiguousarray(values, dtype=np.float32) if x.ndim < 1 or x.size == 0: raise ValueError('Expected a nonempty feature array') width = x.shape[-1] g = np.ones(width, np.float32) if gamma is None else np.ascontiguousarray(gamma, dtype=np.float32) b = np.zeros(width, np.float32) if beta is None else np.ascontiguousarray(beta, dtype=np.float32) if g.shape != (width,) or b.shape != (width,): raise ValueError('gamma and beta must match the last dimension') if x.size > 1048576 or width > 65536: raise ValueError('LayerNorm input exceeds native capacity') output = np.empty_like(x) code = self._layernorm(x.ctypes.data_as(self._ptr), g.ctypes.data_as(self._ptr), b.ctypes.data_as(self._ptr), x.size // width, width, epsilon, output.ctypes.data_as(self._ptr)) if code: raise ValueError(f'Bend rejected LayerNorm inputs (status {code})') return output class BendMatrix: """Resident immutable FP32 projection weights owned by the Bend heap.""" def __init__(self, weights, reducer=None): import numpy as np import threading import weakref self.reducer = reducer or BendReducer() self._lock = threading.RLock() values = np.ascontiguousarray(weights, dtype=np.float32) if values.ndim != 2 or not all(values.shape): raise ValueError('Projection weights must be a nonempty matrix') self.outputs, self.inputs = values.shape lib = self.reducer._lib create = lib.julia_router_matrix_create create.argtypes = [self.reducer._ptr, ctypes.c_uint32, ctypes.c_uint32] create.restype = ctypes.c_void_p release = lib.julia_router_matrix_free release.argtypes = [ctypes.c_void_p] release.restype = None self._linear = lib.julia_router_linear self._linear.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_uint32, ctypes.c_void_p] self._linear.restype = ctypes.c_int self._linear_inference = lib.julia_router_linear_inference self._linear_inference.argtypes = self._linear.argtypes self._linear_inference.restype = ctypes.c_int self._handle = create(values.ctypes.data_as(self.reducer._ptr), self.inputs, self.outputs) if not self._handle: raise ValueError('Bend rejected projection dimensions or nonfinite weights') self._finalize = weakref.finalize(self, release, self._handle) def close(self): with self._lock: self._finalize() def tensor(self, values, *, validate=True): """Fill a contiguous Torch output directly, without NumPy/list/concat wrappers.""" import torch if values.device.type != 'cpu' or values.dtype != torch.float32 or values.shape[-1] != self.inputs: raise ValueError('Bend tensor projection requires matching CPU FP32 inputs') x = values.contiguous() rows = x.numel() // self.inputs if not rows: raise ValueError('Projection input must be nonempty') out = torch.empty((*x.shape[:-1], self.outputs), dtype=x.dtype, device=x.device) linear = self._linear if validate else self._linear_inference limit = min(1048576 // self.inputs, 1048576 // self.outputs) with self._lock: if not self._finalize.alive: raise RuntimeError('Bend matrix has been closed') for start in range(0, rows, limit): code = linear(self._handle, x.data_ptr() + start * self.inputs * 4, min(limit, rows - start), out.data_ptr() + start * self.outputs * 4) if code: raise ValueError(f'Bend projection failed (status {code})') return out def __call__(self, values): import numpy as np x = np.ascontiguousarray(values, dtype=np.float32) if x.ndim < 1 or x.shape[-1] != self.inputs or not x.size: raise ValueError('Projection input must match resident weights') out = np.empty((*x.shape[:-1], self.outputs), dtype=np.float32) with self._lock: if not self._finalize.alive: raise RuntimeError('Bend matrix has been closed') code = self._linear(self._handle, x.ctypes.data_as(self.reducer._ptr), x.size // self.inputs, out.ctypes.data_as(self.reducer._ptr)) if code: raise ValueError(f'Bend projection failed (status {code})') return out