kleeedolinux commited on
Commit ·
81b0424
1
Parent(s): 111207e
Enable native 8192-token Julia-1 inference context
Browse files- README.md +3 -3
- inference-policy.json +3 -2
- julia/data.py +2 -2
- julia/inference.py +12 -2
- julia/router/engine.py +2 -1
- metrics/context-8k-smoke.json +8 -0
- tests/test_context.py +21 -0
README.md
CHANGED
|
@@ -63,7 +63,7 @@ Julia 1 starts from [JHU CLSP's mmBERT-small](https://huggingface.co/jhu-clsp/mm
|
|
| 63 |
| --- | --- | --- |
|
| 64 |
| Main interface | Encoder representations / masked-token modeling | `state` + `question` + 2–20 `options` → one selected option |
|
| 65 |
| Parameters | About 140M | 144.3M including the decision components |
|
| 66 |
-
| Context | Upstream architecture supports up to 8,192 tokens |
|
| 67 |
| Output | Token or encoder features for a downstream task | Scores and a selected answer in the caller's option order |
|
| 68 |
| Usage | General multilingual encoder foundation | Specialized finite-choice decisions |
|
| 69 |
|
|
@@ -92,7 +92,7 @@ engine = load_model(
|
|
| 92 |
"Julia-1",
|
| 93 |
device="cpu",
|
| 94 |
strict_encoding=True,
|
| 95 |
-
max_length=
|
| 96 |
head_length=512,
|
| 97 |
)
|
| 98 |
|
|
@@ -121,7 +121,7 @@ print(result["probabilities"])
|
|
| 121 |
| `options` | 2–20 nonempty answer descriptions, in the order you want returned. |
|
| 122 |
| `type` | `choice` (default), `score` for ordered options, or `noul` for Boolean decisions. |
|
| 123 |
|
| 124 |
-
For `noul`, provide exactly two options: **false first, true second**. For `score`, supply options in their intended order. Strict encoding rejects inputs that exceed the model's limits instead of silently truncating them. The
|
| 125 |
|
| 126 |
The returned percentages are **display values, not calibrated certainty**. If the raw top probability exceeds 95% and every other option is below 4.5%, the result displays 100% for the winner. Values below 1% display as 0%, with their mass redistributed proportionally among the remaining options. The selected index still comes from raw logits. For analysis or your own calibration, use `logits()`.
|
| 127 |
|
|
|
|
| 63 |
| --- | --- | --- |
|
| 64 |
| Main interface | Encoder representations / masked-token modeling | `state` + `question` + 2–20 `options` → one selected option |
|
| 65 |
| Parameters | About 140M | 144.3M including the decision components |
|
| 66 |
+
| Context | Upstream architecture supports up to 8,192 tokens | Runtime supports up to 8,192 combined tokens; historical benchmarks used 1,024 |
|
| 67 |
| Output | Token or encoder features for a downstream task | Scores and a selected answer in the caller's option order |
|
| 68 |
| Usage | General multilingual encoder foundation | Specialized finite-choice decisions |
|
| 69 |
|
|
|
|
| 92 |
"Julia-1",
|
| 93 |
device="cpu",
|
| 94 |
strict_encoding=True,
|
| 95 |
+
max_length=8192,
|
| 96 |
head_length=512,
|
| 97 |
)
|
| 98 |
|
|
|
|
| 121 |
| `options` | 2–20 nonempty answer descriptions, in the order you want returned. |
|
| 122 |
| `type` | `choice` (default), `score` for ordered options, or `noul` for Boolean decisions. |
|
| 123 |
|
| 124 |
+
For `noul`, provide exactly two options: **false first, true second**. For `score`, supply options in their intended order. Strict encoding rejects inputs that exceed the model's limits instead of silently truncating them. The runtime defaults to the checkpoint’s native **8,192-token** total sequence limit, including state, question and options. Historical benchmarks above used 1,024 tokens; An [exact 8,192-token CPU inference smoke test](metrics/context-8k-smoke.json) passed with finite outputs; 8k task accuracy has not been established. The example keeps a 512-token question-and-options budget; each option may use at most 48 tokens.
|
| 125 |
|
| 126 |
The returned percentages are **display values, not calibrated certainty**. If the raw top probability exceeds 95% and every other option is below 4.5%, the result displays 100% for the winner. Values below 1% display as 0%, with their mass redistributed proportionally among the remaining options. The selected index still comes from raw logits. For analysis or your own calibration, use `logits()`.
|
| 127 |
|
inference-policy.json
CHANGED
|
@@ -1,10 +1,11 @@
|
|
| 1 |
{
|
| 2 |
"single_model": true,
|
| 3 |
-
"max_length":
|
| 4 |
"head_length": 512,
|
| 5 |
"strict_encoding": true,
|
| 6 |
"calibration": null,
|
| 7 |
"replacement_qualified": false,
|
| 8 |
"step": 500,
|
| 9 |
-
"weights_sha256": "df853bf7fe424420011f3d0c47a05d7341aa9eefa7fb9f203ea4aada4ad95b72"
|
|
|
|
| 10 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"single_model": true,
|
| 3 |
+
"max_length": 8192,
|
| 4 |
"head_length": 512,
|
| 5 |
"strict_encoding": true,
|
| 6 |
"calibration": null,
|
| 7 |
"replacement_qualified": false,
|
| 8 |
"step": 500,
|
| 9 |
+
"weights_sha256": "df853bf7fe424420011f3d0c47a05d7341aa9eefa7fb9f203ea4aada4ad95b72",
|
| 10 |
+
"context_basis": "Native mmBERT positional limit; long-context task accuracy not established"
|
| 11 |
}
|
julia/data.py
CHANGED
|
@@ -69,7 +69,7 @@ class Decisions(Dataset):
|
|
| 69 |
return self.rows[index]
|
| 70 |
|
| 71 |
|
| 72 |
-
def sequence(tokenizer, row, max_length=
|
| 73 |
if head_length + 4 >= max_length:
|
| 74 |
raise ValueError('max_length must leave room beyond the question head')
|
| 75 |
if any(x is None for x in (tokenizer.mask_token_id, tokenizer.cls_token_id, tokenizer.sep_token_id)):
|
|
@@ -111,7 +111,7 @@ def sequence(tokenizer, row, max_length=1024, head_length=256, *, strict=False):
|
|
| 111 |
|
| 112 |
|
| 113 |
class Collator:
|
| 114 |
-
def __init__(self, tokenizer, max_length=
|
| 115 |
self.tokenizer, self.max_length, self.head_length = tokenizer, max_length, head_length
|
| 116 |
|
| 117 |
def __call__(self, rows, *, include_targets=True):
|
|
|
|
| 69 |
return self.rows[index]
|
| 70 |
|
| 71 |
|
| 72 |
+
def sequence(tokenizer, row, max_length=8192, head_length=256, *, strict=False):
|
| 73 |
if head_length + 4 >= max_length:
|
| 74 |
raise ValueError('max_length must leave room beyond the question head')
|
| 75 |
if any(x is None for x in (tokenizer.mask_token_id, tokenizer.cls_token_id, tokenizer.sep_token_id)):
|
|
|
|
| 111 |
|
| 112 |
|
| 113 |
class Collator:
|
| 114 |
+
def __init__(self, tokenizer, max_length=8192, head_length=256):
|
| 115 |
self.tokenizer, self.max_length, self.head_length = tokenizer, max_length, head_length
|
| 116 |
|
| 117 |
def __call__(self, rows, *, include_targets=True):
|
julia/inference.py
CHANGED
|
@@ -8,12 +8,22 @@ from .data import Collator, validate_row
|
|
| 8 |
from .probabilities import display_probabilities
|
| 9 |
|
| 10 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
class TransformerEngine:
|
| 12 |
-
def __init__(self, checkpoint, device='cuda', max_length=
|
| 13 |
from transformers import AutoModel, AutoTokenizer
|
| 14 |
from .model import JuliaDecisionModel
|
| 15 |
self.device = configure(device)
|
| 16 |
root = Path(checkpoint)
|
|
|
|
| 17 |
if (root / 'INCOMPLETE').exists():
|
| 18 |
raise ValueError('Refusing to load an incomplete INT8 export')
|
| 19 |
self.tokenizer = AutoTokenizer.from_pretrained(root / 'tokenizer', trust_remote_code=False)
|
|
@@ -60,7 +70,7 @@ class TransformerEngine:
|
|
| 60 |
return result
|
| 61 |
|
| 62 |
|
| 63 |
-
def load_model(checkpoint, device='cpu', max_length=
|
| 64 |
"""Load a Julia checkpoint through the supported resident inference runtime."""
|
| 65 |
from .router.engine import FastEngine
|
| 66 |
if checkpoint is None:
|
|
|
|
| 8 |
from .probabilities import display_probabilities
|
| 9 |
|
| 10 |
|
| 11 |
+
def context_length(checkpoint, requested):
|
| 12 |
+
config = json.loads((Path(checkpoint) / 'encoder/config.json').read_text())
|
| 13 |
+
limit = config['max_position_embeddings']
|
| 14 |
+
value = limit if requested is None else requested
|
| 15 |
+
if type(value) is not int or not 1 <= value <= limit:
|
| 16 |
+
raise ValueError(f'max_length must be an integer between 1 and {limit}')
|
| 17 |
+
return value
|
| 18 |
+
|
| 19 |
+
|
| 20 |
class TransformerEngine:
|
| 21 |
+
def __init__(self, checkpoint, device='cuda', max_length=None, head_length=256, *, memory_map=True):
|
| 22 |
from transformers import AutoModel, AutoTokenizer
|
| 23 |
from .model import JuliaDecisionModel
|
| 24 |
self.device = configure(device)
|
| 25 |
root = Path(checkpoint)
|
| 26 |
+
max_length = context_length(root, max_length)
|
| 27 |
if (root / 'INCOMPLETE').exists():
|
| 28 |
raise ValueError('Refusing to load an incomplete INT8 export')
|
| 29 |
self.tokenizer = AutoTokenizer.from_pretrained(root / 'tokenizer', trust_remote_code=False)
|
|
|
|
| 70 |
return result
|
| 71 |
|
| 72 |
|
| 73 |
+
def load_model(checkpoint, device='cpu', max_length=None, head_length=256, *, backend=None, **kwargs):
|
| 74 |
"""Load a Julia checkpoint through the supported resident inference runtime."""
|
| 75 |
from .router.engine import FastEngine
|
| 76 |
if checkpoint is None:
|
julia/router/engine.py
CHANGED
|
@@ -38,7 +38,7 @@ class FastEngine(Engine):
|
|
| 38 |
Optional torch.compile specializes the transformer, while Bend handles CPU
|
| 39 |
softmax/selection via ctypes. CUDA softmax stays on-device to avoid a roundtrip.
|
| 40 |
"""
|
| 41 |
-
def __init__(self, checkpoint, device='cpu', max_length=
|
| 42 |
batch_size=16, encoding_cache=2048, token_cache=8192,
|
| 43 |
compile_model=False, library=None, bend_postprocess=False, transformer_backend=None,
|
| 44 |
strict_encoding=False, marker_only_head=None, memory_map=True, padding_ratio=1.25):
|
|
@@ -60,6 +60,7 @@ class FastEngine(Engine):
|
|
| 60 |
self.model.marker_only_head = (self.device.type == 'cpu' if marker_only_head is None else marker_only_head)
|
| 61 |
self.strict_encoding = strict_encoding
|
| 62 |
self.batch_size = batch_size
|
|
|
|
| 63 |
self.max_length, self.head_length = max_length, head_length
|
| 64 |
self.encoding_cache = encoding_cache
|
| 65 |
self._encoded = OrderedDict()
|
|
|
|
| 38 |
Optional torch.compile specializes the transformer, while Bend handles CPU
|
| 39 |
softmax/selection via ctypes. CUDA softmax stays on-device to avoid a roundtrip.
|
| 40 |
"""
|
| 41 |
+
def __init__(self, checkpoint, device='cpu', max_length=None, head_length=256,
|
| 42 |
batch_size=16, encoding_cache=2048, token_cache=8192,
|
| 43 |
compile_model=False, library=None, bend_postprocess=False, transformer_backend=None,
|
| 44 |
strict_encoding=False, marker_only_head=None, memory_map=True, padding_ratio=1.25):
|
|
|
|
| 60 |
self.model.marker_only_head = (self.device.type == 'cpu' if marker_only_head is None else marker_only_head)
|
| 61 |
self.strict_encoding = strict_encoding
|
| 62 |
self.batch_size = batch_size
|
| 63 |
+
max_length = self.collate.max_length
|
| 64 |
self.max_length, self.head_length = max_length, head_length
|
| 65 |
self.encoding_cache = encoding_cache
|
| 66 |
self._encoded = OrderedDict()
|
metrics/context-8k-smoke.json
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"tokens": 8192,
|
| 3 |
+
"device": "cpu",
|
| 4 |
+
"elapsed_seconds": 24.356244013994,
|
| 5 |
+
"finite_logits": true,
|
| 6 |
+
"max_length": 8192,
|
| 7 |
+
"scope": "Runtime execution smoke; not long-context accuracy evaluation"
|
| 8 |
+
}
|
tests/test_context.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import unittest
|
| 2 |
+
from pathlib import Path
|
| 3 |
+
from julia.inference import context_length
|
| 4 |
+
from julia.data import sequence
|
| 5 |
+
|
| 6 |
+
class Tokenizer:
|
| 7 |
+
mask_token='[MASK]';mask_token_id=4;cls_token_id=1;sep_token_id=2
|
| 8 |
+
def __call__(self,text,**kwargs):return {'input_ids':[5]*len(text.split())}
|
| 9 |
+
|
| 10 |
+
class ContextTests(unittest.TestCase):
|
| 11 |
+
def test_default_uses_native_checkpoint_limit(self):
|
| 12 |
+
self.assertEqual(context_length(Path(__file__).resolve().parents[1],None),8192)
|
| 13 |
+
def test_rejects_beyond_native_limit(self):
|
| 14 |
+
with self.assertRaises(ValueError):context_length(Path(__file__).resolve().parents[1],8193)
|
| 15 |
+
def test_full_budget_and_strict_overflow(self):
|
| 16 |
+
row=dict(state='',question='choose',options=['a','b'])
|
| 17 |
+
overhead=len(sequence(Tokenizer(),row,strict=True)['ids'])
|
| 18 |
+
row['state']=' '.join(['x']*(8192-overhead))
|
| 19 |
+
self.assertEqual(len(sequence(Tokenizer(),row,strict=True)['ids']),8192)
|
| 20 |
+
row['state']+=' x'
|
| 21 |
+
with self.assertRaises(ValueError):sequence(Tokenizer(),row,strict=True)
|