kleeedolinux commited on
Commit
81b0424
·
1 Parent(s): 111207e

Enable native 8192-token Julia-1 inference context

Browse files
README.md CHANGED
@@ -63,7 +63,7 @@ Julia 1 starts from [JHU CLSP's mmBERT-small](https://huggingface.co/jhu-clsp/mm
63
  | --- | --- | --- |
64
  | Main interface | Encoder representations / masked-token modeling | `state` + `question` + 2–20 `options` → one selected option |
65
  | Parameters | About 140M | 144.3M including the decision components |
66
- | Context | Upstream architecture supports up to 8,192 tokens | Evaluated Julia decision path uses up to 1,024 tokens |
67
  | Output | Token or encoder features for a downstream task | Scores and a selected answer in the caller's option order |
68
  | Usage | General multilingual encoder foundation | Specialized finite-choice decisions |
69
 
@@ -92,7 +92,7 @@ engine = load_model(
92
  "Julia-1",
93
  device="cpu",
94
  strict_encoding=True,
95
- max_length=1024,
96
  head_length=512,
97
  )
98
 
@@ -121,7 +121,7 @@ print(result["probabilities"])
121
  | `options` | 2–20 nonempty answer descriptions, in the order you want returned. |
122
  | `type` | `choice` (default), `score` for ordered options, or `noul` for Boolean decisions. |
123
 
124
- For `noul`, provide exactly two options: **false first, true second**. For `score`, supply options in their intended order. Strict encoding rejects inputs that exceed the model's limits instead of silently truncating them. The evaluated configuration uses a 1,024-token total sequence and a 512-token question-and-options budget; each option may use at most 48 tokens.
125
 
126
  The returned percentages are **display values, not calibrated certainty**. If the raw top probability exceeds 95% and every other option is below 4.5%, the result displays 100% for the winner. Values below 1% display as 0%, with their mass redistributed proportionally among the remaining options. The selected index still comes from raw logits. For analysis or your own calibration, use `logits()`.
127
 
 
63
  | --- | --- | --- |
64
  | Main interface | Encoder representations / masked-token modeling | `state` + `question` + 2–20 `options` → one selected option |
65
  | Parameters | About 140M | 144.3M including the decision components |
66
+ | Context | Upstream architecture supports up to 8,192 tokens | Runtime supports up to 8,192 combined tokens; historical benchmarks used 1,024 |
67
  | Output | Token or encoder features for a downstream task | Scores and a selected answer in the caller's option order |
68
  | Usage | General multilingual encoder foundation | Specialized finite-choice decisions |
69
 
 
92
  "Julia-1",
93
  device="cpu",
94
  strict_encoding=True,
95
+ max_length=8192,
96
  head_length=512,
97
  )
98
 
 
121
  | `options` | 2–20 nonempty answer descriptions, in the order you want returned. |
122
  | `type` | `choice` (default), `score` for ordered options, or `noul` for Boolean decisions. |
123
 
124
+ For `noul`, provide exactly two options: **false first, true second**. For `score`, supply options in their intended order. Strict encoding rejects inputs that exceed the model's limits instead of silently truncating them. The runtime defaults to the checkpoint’s native **8,192-token** total sequence limit, including state, question and options. Historical benchmarks above used 1,024 tokens; An [exact 8,192-token CPU inference smoke test](metrics/context-8k-smoke.json) passed with finite outputs; 8k task accuracy has not been established. The example keeps a 512-token question-and-options budget; each option may use at most 48 tokens.
125
 
126
  The returned percentages are **display values, not calibrated certainty**. If the raw top probability exceeds 95% and every other option is below 4.5%, the result displays 100% for the winner. Values below 1% display as 0%, with their mass redistributed proportionally among the remaining options. The selected index still comes from raw logits. For analysis or your own calibration, use `logits()`.
127
 
inference-policy.json CHANGED
@@ -1,10 +1,11 @@
1
  {
2
  "single_model": true,
3
- "max_length": 1024,
4
  "head_length": 512,
5
  "strict_encoding": true,
6
  "calibration": null,
7
  "replacement_qualified": false,
8
  "step": 500,
9
- "weights_sha256": "df853bf7fe424420011f3d0c47a05d7341aa9eefa7fb9f203ea4aada4ad95b72"
 
10
  }
 
1
  {
2
  "single_model": true,
3
+ "max_length": 8192,
4
  "head_length": 512,
5
  "strict_encoding": true,
6
  "calibration": null,
7
  "replacement_qualified": false,
8
  "step": 500,
9
+ "weights_sha256": "df853bf7fe424420011f3d0c47a05d7341aa9eefa7fb9f203ea4aada4ad95b72",
10
+ "context_basis": "Native mmBERT positional limit; long-context task accuracy not established"
11
  }
julia/data.py CHANGED
@@ -69,7 +69,7 @@ class Decisions(Dataset):
69
  return self.rows[index]
70
 
71
 
72
- def sequence(tokenizer, row, max_length=1024, head_length=256, *, strict=False):
73
  if head_length + 4 >= max_length:
74
  raise ValueError('max_length must leave room beyond the question head')
75
  if any(x is None for x in (tokenizer.mask_token_id, tokenizer.cls_token_id, tokenizer.sep_token_id)):
@@ -111,7 +111,7 @@ def sequence(tokenizer, row, max_length=1024, head_length=256, *, strict=False):
111
 
112
 
113
  class Collator:
114
- def __init__(self, tokenizer, max_length=512, head_length=256):
115
  self.tokenizer, self.max_length, self.head_length = tokenizer, max_length, head_length
116
 
117
  def __call__(self, rows, *, include_targets=True):
 
69
  return self.rows[index]
70
 
71
 
72
+ def sequence(tokenizer, row, max_length=8192, head_length=256, *, strict=False):
73
  if head_length + 4 >= max_length:
74
  raise ValueError('max_length must leave room beyond the question head')
75
  if any(x is None for x in (tokenizer.mask_token_id, tokenizer.cls_token_id, tokenizer.sep_token_id)):
 
111
 
112
 
113
  class Collator:
114
+ def __init__(self, tokenizer, max_length=8192, head_length=256):
115
  self.tokenizer, self.max_length, self.head_length = tokenizer, max_length, head_length
116
 
117
  def __call__(self, rows, *, include_targets=True):
julia/inference.py CHANGED
@@ -8,12 +8,22 @@ from .data import Collator, validate_row
8
  from .probabilities import display_probabilities
9
 
10
 
 
 
 
 
 
 
 
 
 
11
  class TransformerEngine:
12
- def __init__(self, checkpoint, device='cuda', max_length=1024, head_length=256, *, memory_map=True):
13
  from transformers import AutoModel, AutoTokenizer
14
  from .model import JuliaDecisionModel
15
  self.device = configure(device)
16
  root = Path(checkpoint)
 
17
  if (root / 'INCOMPLETE').exists():
18
  raise ValueError('Refusing to load an incomplete INT8 export')
19
  self.tokenizer = AutoTokenizer.from_pretrained(root / 'tokenizer', trust_remote_code=False)
@@ -60,7 +70,7 @@ class TransformerEngine:
60
  return result
61
 
62
 
63
- def load_model(checkpoint, device='cpu', max_length=1024, head_length=256, *, backend=None, **kwargs):
64
  """Load a Julia checkpoint through the supported resident inference runtime."""
65
  from .router.engine import FastEngine
66
  if checkpoint is None:
 
8
  from .probabilities import display_probabilities
9
 
10
 
11
+ def context_length(checkpoint, requested):
12
+ config = json.loads((Path(checkpoint) / 'encoder/config.json').read_text())
13
+ limit = config['max_position_embeddings']
14
+ value = limit if requested is None else requested
15
+ if type(value) is not int or not 1 <= value <= limit:
16
+ raise ValueError(f'max_length must be an integer between 1 and {limit}')
17
+ return value
18
+
19
+
20
  class TransformerEngine:
21
+ def __init__(self, checkpoint, device='cuda', max_length=None, head_length=256, *, memory_map=True):
22
  from transformers import AutoModel, AutoTokenizer
23
  from .model import JuliaDecisionModel
24
  self.device = configure(device)
25
  root = Path(checkpoint)
26
+ max_length = context_length(root, max_length)
27
  if (root / 'INCOMPLETE').exists():
28
  raise ValueError('Refusing to load an incomplete INT8 export')
29
  self.tokenizer = AutoTokenizer.from_pretrained(root / 'tokenizer', trust_remote_code=False)
 
70
  return result
71
 
72
 
73
+ def load_model(checkpoint, device='cpu', max_length=None, head_length=256, *, backend=None, **kwargs):
74
  """Load a Julia checkpoint through the supported resident inference runtime."""
75
  from .router.engine import FastEngine
76
  if checkpoint is None:
julia/router/engine.py CHANGED
@@ -38,7 +38,7 @@ class FastEngine(Engine):
38
  Optional torch.compile specializes the transformer, while Bend handles CPU
39
  softmax/selection via ctypes. CUDA softmax stays on-device to avoid a roundtrip.
40
  """
41
- def __init__(self, checkpoint, device='cpu', max_length=1024, head_length=256,
42
  batch_size=16, encoding_cache=2048, token_cache=8192,
43
  compile_model=False, library=None, bend_postprocess=False, transformer_backend=None,
44
  strict_encoding=False, marker_only_head=None, memory_map=True, padding_ratio=1.25):
@@ -60,6 +60,7 @@ class FastEngine(Engine):
60
  self.model.marker_only_head = (self.device.type == 'cpu' if marker_only_head is None else marker_only_head)
61
  self.strict_encoding = strict_encoding
62
  self.batch_size = batch_size
 
63
  self.max_length, self.head_length = max_length, head_length
64
  self.encoding_cache = encoding_cache
65
  self._encoded = OrderedDict()
 
38
  Optional torch.compile specializes the transformer, while Bend handles CPU
39
  softmax/selection via ctypes. CUDA softmax stays on-device to avoid a roundtrip.
40
  """
41
+ def __init__(self, checkpoint, device='cpu', max_length=None, head_length=256,
42
  batch_size=16, encoding_cache=2048, token_cache=8192,
43
  compile_model=False, library=None, bend_postprocess=False, transformer_backend=None,
44
  strict_encoding=False, marker_only_head=None, memory_map=True, padding_ratio=1.25):
 
60
  self.model.marker_only_head = (self.device.type == 'cpu' if marker_only_head is None else marker_only_head)
61
  self.strict_encoding = strict_encoding
62
  self.batch_size = batch_size
63
+ max_length = self.collate.max_length
64
  self.max_length, self.head_length = max_length, head_length
65
  self.encoding_cache = encoding_cache
66
  self._encoded = OrderedDict()
metrics/context-8k-smoke.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "tokens": 8192,
3
+ "device": "cpu",
4
+ "elapsed_seconds": 24.356244013994,
5
+ "finite_logits": true,
6
+ "max_length": 8192,
7
+ "scope": "Runtime execution smoke; not long-context accuracy evaluation"
8
+ }
tests/test_context.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import unittest
2
+ from pathlib import Path
3
+ from julia.inference import context_length
4
+ from julia.data import sequence
5
+
6
+ class Tokenizer:
7
+ mask_token='[MASK]';mask_token_id=4;cls_token_id=1;sep_token_id=2
8
+ def __call__(self,text,**kwargs):return {'input_ids':[5]*len(text.split())}
9
+
10
+ class ContextTests(unittest.TestCase):
11
+ def test_default_uses_native_checkpoint_limit(self):
12
+ self.assertEqual(context_length(Path(__file__).resolve().parents[1],None),8192)
13
+ def test_rejects_beyond_native_limit(self):
14
+ with self.assertRaises(ValueError):context_length(Path(__file__).resolve().parents[1],8193)
15
+ def test_full_budget_and_strict_overflow(self):
16
+ row=dict(state='',question='choose',options=['a','b'])
17
+ overhead=len(sequence(Tokenizer(),row,strict=True)['ids'])
18
+ row['state']=' '.join(['x']*(8192-overhead))
19
+ self.assertEqual(len(sequence(Tokenizer(),row,strict=True)['ids']),8192)
20
+ row['state']+=' x'
21
+ with self.assertRaises(ValueError):sequence(Tokenizer(),row,strict=True)