Julia-1 / julia /router /native.py
kleeedolinux
Publish Julia 1 model and Python runtime
5278d6b
Raw
History Blame Contribute Delete
7.84 kB
"""ctypes boundary; no subprocess or compilation in the request path."""
import ctypes
import math
import os
from pathlib import Path
class BendReducer:
def __init__(self, library=None):
path = Path(library or os.environ.get('JULIA_ROUTER_LIBRARY') or
Path(__file__).parent / 'build/libjulia_router.so')
try:
self._lib = ctypes.CDLL(str(path.resolve()))
except OSError as error:
raise RuntimeError(
f'Cannot load Bend runtime {path.resolve()}. From Julia-1 run '
'python -m julia.router.build --native-cpu; or explicitly select '
'transformer_backend="torch".') from error
self._argmax = self._lib.julia_router_argmax
self._argmax.argtypes = [ctypes.POINTER(ctypes.c_float), ctypes.c_uint32,
ctypes.POINTER(ctypes.c_uint32)]
self._argmax.restype = ctypes.c_int
self._ptr = ctypes.POINTER(ctypes.c_float)
self._softmax = self._lib.julia_router_softmax
self._softmax.argtypes = [self._ptr, ctypes.c_uint32, self._ptr, ctypes.POINTER(ctypes.c_uint32)]
self._softmax.restype = ctypes.c_int
self._layernorm = self._lib.julia_router_layernorm
self._layernorm.argtypes = [self._ptr, self._ptr, self._ptr, ctypes.c_uint32,
ctypes.c_uint32, ctypes.c_float, self._ptr]
self._layernorm.restype = ctypes.c_int
def profile(self, *, enabled=False, reset=False):
"""Process-wide packed projection timings; opt in only for diagnostics."""
fn = self._lib.julia_router_profile
fn.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.POINTER(ctypes.c_double)]
fn.restype = None
values = (ctypes.c_double * 4)()
fn(enabled, reset, values)
return dict(zip(('pack_ms', 'execute_ms', 'unpack_ms', 'calls'), values))
def argmax(self, scores):
scores = tuple(scores)
if not 1 <= len(scores) <= 1048576:
raise ValueError('Expected 1–1048576 scores')
if not all(math.isfinite(x) for x in scores):
raise ValueError('Scores must be finite')
values = (ctypes.c_float * len(scores))(*scores)
result = ctypes.c_uint32()
code = self._argmax(values, len(scores), ctypes.byref(result))
if code:
raise ValueError(f'Bend rejected scores (status {code})')
return result.value
def softmax(self, scores):
import numpy as np
values = np.ascontiguousarray(scores, dtype=np.float32)
if values.ndim != 1 or not 1 <= values.size <= 1048576:
raise ValueError('Expected a nonempty vector of at most 1048576 scores')
output = np.empty_like(values)
index = ctypes.c_uint32()
code = self._softmax(values.ctypes.data_as(self._ptr), values.size,
output.ctypes.data_as(self._ptr), ctypes.byref(index))
if code:
raise ValueError(f'Bend rejected scores (status {code})')
return index.value, output
def layernorm(self, values, gamma=None, beta=None, epsilon=1e-5):
import numpy as np
x = np.ascontiguousarray(values, dtype=np.float32)
if x.ndim < 1 or x.size == 0:
raise ValueError('Expected a nonempty feature array')
width = x.shape[-1]
g = np.ones(width, np.float32) if gamma is None else np.ascontiguousarray(gamma, dtype=np.float32)
b = np.zeros(width, np.float32) if beta is None else np.ascontiguousarray(beta, dtype=np.float32)
if g.shape != (width,) or b.shape != (width,):
raise ValueError('gamma and beta must match the last dimension')
if x.size > 1048576 or width > 65536:
raise ValueError('LayerNorm input exceeds native capacity')
output = np.empty_like(x)
code = self._layernorm(x.ctypes.data_as(self._ptr), g.ctypes.data_as(self._ptr),
b.ctypes.data_as(self._ptr), x.size // width, width, epsilon,
output.ctypes.data_as(self._ptr))
if code:
raise ValueError(f'Bend rejected LayerNorm inputs (status {code})')
return output
class BendMatrix:
"""Resident immutable FP32 projection weights owned by the Bend heap."""
def __init__(self, weights, reducer=None):
import numpy as np
import threading
import weakref
self.reducer = reducer or BendReducer()
self._lock = threading.RLock()
values = np.ascontiguousarray(weights, dtype=np.float32)
if values.ndim != 2 or not all(values.shape):
raise ValueError('Projection weights must be a nonempty matrix')
self.outputs, self.inputs = values.shape
lib = self.reducer._lib
create = lib.julia_router_matrix_create
create.argtypes = [self.reducer._ptr, ctypes.c_uint32, ctypes.c_uint32]
create.restype = ctypes.c_void_p
release = lib.julia_router_matrix_free
release.argtypes = [ctypes.c_void_p]
release.restype = None
self._linear = lib.julia_router_linear
self._linear.argtypes = [ctypes.c_void_p, ctypes.c_void_p, ctypes.c_uint32, ctypes.c_void_p]
self._linear.restype = ctypes.c_int
self._linear_inference = lib.julia_router_linear_inference
self._linear_inference.argtypes = self._linear.argtypes
self._linear_inference.restype = ctypes.c_int
self._handle = create(values.ctypes.data_as(self.reducer._ptr), self.inputs, self.outputs)
if not self._handle:
raise ValueError('Bend rejected projection dimensions or nonfinite weights')
self._finalize = weakref.finalize(self, release, self._handle)
def close(self):
with self._lock:
self._finalize()
def tensor(self, values, *, validate=True):
"""Fill a contiguous Torch output directly, without NumPy/list/concat wrappers."""
import torch
if values.device.type != 'cpu' or values.dtype != torch.float32 or values.shape[-1] != self.inputs:
raise ValueError('Bend tensor projection requires matching CPU FP32 inputs')
x = values.contiguous()
rows = x.numel() // self.inputs
if not rows:
raise ValueError('Projection input must be nonempty')
out = torch.empty((*x.shape[:-1], self.outputs), dtype=x.dtype, device=x.device)
linear = self._linear if validate else self._linear_inference
limit = min(1048576 // self.inputs, 1048576 // self.outputs)
with self._lock:
if not self._finalize.alive:
raise RuntimeError('Bend matrix has been closed')
for start in range(0, rows, limit):
code = linear(self._handle,
x.data_ptr() + start * self.inputs * 4,
min(limit, rows - start),
out.data_ptr() + start * self.outputs * 4)
if code:
raise ValueError(f'Bend projection failed (status {code})')
return out
def __call__(self, values):
import numpy as np
x = np.ascontiguousarray(values, dtype=np.float32)
if x.ndim < 1 or x.shape[-1] != self.inputs or not x.size:
raise ValueError('Projection input must match resident weights')
out = np.empty((*x.shape[:-1], self.outputs), dtype=np.float32)
with self._lock:
if not self._finalize.alive:
raise RuntimeError('Bend matrix has been closed')
code = self._linear(self._handle, x.ctypes.data_as(self.reducer._ptr),
x.size // self.inputs, out.ctypes.data_as(self.reducer._ptr))
if code:
raise ValueError(f'Bend projection failed (status {code})')
return out