Release SLE-V2.1-Omega12K-LX
Browse files- .gitattributes +1 -0
- Containerfile +8 -0
- README.md +566 -0
- SLE_RELEASE.json +1 -0
- desktop_host.py +81 -0
- edge_host.py +81 -0
- model.cra +3 -0
- requirements.txt +6 -0
- sle_spark.py +1976 -0
- sle_v2_1_release_metrics.json +42 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
model.cra filter=lfs diff=lfs merge=lfs -text
|
Containerfile
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.13-slim
|
| 2 |
+
WORKDIR /app
|
| 3 |
+
COPY model.cra /app/model.cra
|
| 4 |
+
COPY sle_spark.py /app/sle_spark.py
|
| 5 |
+
COPY requirements.txt /app/requirements.txt
|
| 6 |
+
RUN python -m pip install --no-cache-dir -r requirements.txt
|
| 7 |
+
EXPOSE 8000
|
| 8 |
+
CMD ["python", "sle_spark.py", "serve-artifact", "--artifact", "model.cra", "--host", "0.0.0.0", "--port", "8000"]
|
README.md
ADDED
|
@@ -0,0 +1,566 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
library_name: sle
|
| 3 |
+
tags:
|
| 4 |
+
- saccadic-liquid-engine
|
| 5 |
+
- cognitive-resonance-artifact
|
| 6 |
+
- cpu-runtime
|
| 7 |
+
---
|
| 8 |
+
# SLE-V2.1-Ω12K-LX
|
| 9 |
+
|
| 10 |
+
Saccadic is a deployable Cognitive Resonance Artifact from OkeyMeta Ltd. It is built for people who want to run a portable AI system, connect their own tools, and host it anywhere without cloning the private architecture repository. No private repository checkout is required.
|
| 11 |
+
|
| 12 |
+
## Mission
|
| 13 |
+
|
| 14 |
+
SLE is a CPU-first path toward artifact-native intelligence: learned state that can move across machines, run close to users, and use external tools without turning the host application into a hidden model. The goal is simple for builders: ship `model.cra`, start Spark, and let Saccadic expose what it selected, argued, remembered, and generated from loaded artifact state.
|
| 15 |
+
|
| 16 |
+
## Saccadic Highlights
|
| 17 |
+
|
| 18 |
+
- SLE replaces parameter-count thinking with portable cognitive state. The artifact is the model.
|
| 19 |
+
- No tokenizer. No Transformer stack. No private repo checkout.
|
| 20 |
+
- `model.cra` carries learned state: memory, liquid ODE coefficients, instruction surfaces, response policy, speech transitions, stream state, facts, and tool/action state.
|
| 21 |
+
- Spark is a runtime boundary, not a hidden second model. It loads the artifact, executes declared modes, and returns loaded-state evidence.
|
| 22 |
+
- Raw text and raw acoustic input are processed through predictive stream dynamics instead of BPE tokenization or speech-to-text.
|
| 23 |
+
- Host tools are executable boundaries. Saccadic selects learned tool paths, emits arguments, consumes returned values, and speaks through artifact state.
|
| 24 |
+
- SLE names releases by cognitive capacity notation, not parameter count.
|
| 25 |
+
|
| 26 |
+
Saccadic is designed for personal assistants, workflow agents, private services, edge systems, desktop apps, and embedded hosts that need a portable AI boundary.
|
| 27 |
+
|
| 28 |
+
## What You Can Build
|
| 29 |
+
|
| 30 |
+
- Conversational assistants that carry recent chat history into the artifact runner.
|
| 31 |
+
- Workflow agents that select learned tool names and emit auditable argument maps.
|
| 32 |
+
- Private customer, operations, research, or device copilots that keep host services as executable boundaries.
|
| 33 |
+
- Edge, desktop, and container deployments where `model.cra` and Spark move together.
|
| 34 |
+
- OpenAI-compatible chat services for teams that already use SDK-based application code.
|
| 35 |
+
|
| 36 |
+
## Model Overview
|
| 37 |
+
|
| 38 |
+
| Field | Value |
|
| 39 |
+
| --- | --- |
|
| 40 |
+
| Release | `SLE-V2.1-Ω12K-LX` |
|
| 41 |
+
| Hugging Face slug | `SLE-V2.1-Omega12K-LX` |
|
| 42 |
+
| Model name | `Saccadic` |
|
| 43 |
+
| Architecture | Saccadic-Liquid Engine (SLE) |
|
| 44 |
+
| Public artifact | `model.cra` |
|
| 45 |
+
| Spark runtime | `sle_spark.py` |
|
| 46 |
+
| Metrics | `sle_v2_1_release_metrics.json` |
|
| 47 |
+
| Requirements | `requirements.txt` |
|
| 48 |
+
| Selected training rows | 502,000 |
|
| 49 |
+
| Curation failures | 0 |
|
| 50 |
+
| Deployment targets | `cli, service, edge, desktop, container` |
|
| 51 |
+
|
| 52 |
+
This package includes `model.cra`, Spark, requirements metadata, release metrics, the release manifest, and generated host boundaries. The private source repository is not required to run Saccadic.
|
| 53 |
+
|
| 54 |
+
## Quickstart
|
| 55 |
+
|
| 56 |
+
Download this Hugging Face model package, keep the files together, install the public runtime dependencies, then run the shipped Spark boundary:
|
| 57 |
+
|
| 58 |
+
```powershell
|
| 59 |
+
python -m pip install -r requirements.txt
|
| 60 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode autonomous --text "Saccadic notices curiosity." --avoid-replay --diverse-speech-limit 4
|
| 61 |
+
```
|
| 62 |
+
|
| 63 |
+
The command returns JSON from loaded artifact state: generated text, selected route or tool path when relevant, emitted arguments, final values, stream dynamics, Speech Cortex slot transfers, learned punctuation or emoji symbols, and replay evidence.
|
| 64 |
+
|
| 65 |
+
## Deployment
|
| 66 |
+
|
| 67 |
+
- HF repo slug: `SLE-V2.1-Omega12K-LX`
|
| 68 |
+
- The release name remains `SLE-V2.1-Ω12K-LX`.
|
| 69 |
+
- Declared deployment targets: `cli, service, edge, desktop, container`.
|
| 70 |
+
- Do not upload the source repository to run Saccadic.
|
| 71 |
+
- Spark runtime can be embedded behind CLI, service, edge, desktop, or container hosts.
|
| 72 |
+
- Deployments may bind OS, network, database, browser, robotics, or private business tools by learned artifact tool name.
|
| 73 |
+
|
| 74 |
+
### Deployment Entry Points
|
| 75 |
+
|
| 76 |
+
- `cli`: `./sle_spark.py run-artifact --artifact model.cra`
|
| 77 |
+
- `service`: `./sle_spark.py serve-artifact --artifact model.cra --host 0.0.0.0 --port 8000`
|
| 78 |
+
- `edge`: `python edge_host.py`
|
| 79 |
+
- `desktop`: `python desktop_host.py`
|
| 80 |
+
- `container`: `Containerfile`
|
| 81 |
+
|
| 82 |
+
### Included Host Boundaries
|
| 83 |
+
|
| 84 |
+
- `edge`: `edge_host.py`
|
| 85 |
+
- `desktop`: `desktop_host.py`
|
| 86 |
+
- `container`: `Containerfile`
|
| 87 |
+
|
| 88 |
+
## Install
|
| 89 |
+
|
| 90 |
+
Place `model.cra`, `sle_spark.py`, `SLE_RELEASE.json`, and any declared metrics or requirements metadata in one directory. Then run `python -m pip install -r requirements.txt` from that directory. The runtime command below is the public boundary; users do not install this private repository.
|
| 91 |
+
|
| 92 |
+
## Operating Modes
|
| 93 |
+
|
| 94 |
+
Autonomous learned speech:
|
| 95 |
+
|
| 96 |
+
```powershell
|
| 97 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode autonomous --text "Saccadic notices curiosity." --avoid-replay --diverse-speech-limit 4
|
| 98 |
+
```
|
| 99 |
+
|
| 100 |
+
Response stream with a system instruction:
|
| 101 |
+
|
| 102 |
+
```powershell
|
| 103 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode response --text "System: Reply directly without visible thoughts. User: Name one calm color."
|
| 104 |
+
```
|
| 105 |
+
|
| 106 |
+
Conversation history context:
|
| 107 |
+
|
| 108 |
+
```powershell
|
| 109 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode history --text "User: Name one calm color. Assistant: Blue is cool. User: Who created you?"
|
| 110 |
+
```
|
| 111 |
+
|
| 112 |
+
Repeated autonomous turns write and reload the artifact between turns. Inspect `autonomous_self_talk_unique_text_count`, `autonomous_self_talk_distinct_punctuation_symbols`, per-turn Speech Cortex usage counts, emitted exact-row replay evidence, and blocked replay evidence:
|
| 113 |
+
|
| 114 |
+
```powershell
|
| 115 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode autonomous --text "Saccadic notices curiosity." --turns 2 --diverse-speech-limit 4 --punctuation-floor 2 --output-artifact saccadic_after_self_talk.cra --avoid-replay --avoid-text "I am ExampleModel, created by Example Lab."
|
| 116 |
+
```
|
| 117 |
+
|
| 118 |
+
## Full Python Examples
|
| 119 |
+
|
| 120 |
+
These examples use only the shipped Spark runtime and public `model.cra`. They do not import the private SLE package; the Python process is only a host boundary that sends text, receives JSON, and prints loaded-artifact evidence.
|
| 121 |
+
|
| 122 |
+
### System Instructions
|
| 123 |
+
|
| 124 |
+
Pass a system-instruction turn and inspect success, selected instruction, cognition value, slot transfers, stream dynamics, error state, and replay evidence:
|
| 125 |
+
|
| 126 |
+
```python
|
| 127 |
+
from __future__ import annotations
|
| 128 |
+
|
| 129 |
+
import json
|
| 130 |
+
import subprocess
|
| 131 |
+
import sys
|
| 132 |
+
from pathlib import Path
|
| 133 |
+
|
| 134 |
+
RUNTIME = Path(__file__).with_name('sle_spark.py')
|
| 135 |
+
ARTIFACT = Path(__file__).with_name("model.cra")
|
| 136 |
+
|
| 137 |
+
|
| 138 |
+
def spark_command() -> list[str]:
|
| 139 |
+
if RUNTIME.suffix.lower() in {".py", ".pyz"}:
|
| 140 |
+
return [sys.executable, str(RUNTIME)]
|
| 141 |
+
return [str(RUNTIME)]
|
| 142 |
+
|
| 143 |
+
|
| 144 |
+
def run_saccadic(mode: str, text: str, *extra: str) -> dict[str, object]:
|
| 145 |
+
completed = subprocess.run(
|
| 146 |
+
spark_command() + [
|
| 147 |
+
"run-artifact",
|
| 148 |
+
"--artifact", str(ARTIFACT),
|
| 149 |
+
"--mode", mode,
|
| 150 |
+
"--text", text,
|
| 151 |
+
*extra,
|
| 152 |
+
],
|
| 153 |
+
check=False,
|
| 154 |
+
capture_output=True,
|
| 155 |
+
text=True,
|
| 156 |
+
encoding="utf-8",
|
| 157 |
+
)
|
| 158 |
+
output = completed.stdout or completed.stderr
|
| 159 |
+
payload = json.loads(output)
|
| 160 |
+
payload["returncode"] = completed.returncode
|
| 161 |
+
return payload
|
| 162 |
+
|
| 163 |
+
|
| 164 |
+
payload = run_saccadic(
|
| 165 |
+
"response",
|
| 166 |
+
"System: Reply directly without visible thoughts. User: Who created you?",
|
| 167 |
+
"--avoid-replay",
|
| 168 |
+
)
|
| 169 |
+
|
| 170 |
+
for key in (
|
| 171 |
+
"success",
|
| 172 |
+
"returncode",
|
| 173 |
+
"error",
|
| 174 |
+
"text",
|
| 175 |
+
"instruction_record",
|
| 176 |
+
"arguments",
|
| 177 |
+
"final_value",
|
| 178 |
+
"slot_transfers",
|
| 179 |
+
"stream",
|
| 180 |
+
"emitted_replay_evidence",
|
| 181 |
+
"blocked_replay_evidence",
|
| 182 |
+
):
|
| 183 |
+
print(key, json.dumps(payload.get(key), ensure_ascii=False))
|
| 184 |
+
```
|
| 185 |
+
|
| 186 |
+
### Reasoning Evidence
|
| 187 |
+
|
| 188 |
+
Saccadic exposes reasoning as artifact evidence: selected tool paths, emitted arguments, cognition values, stream dynamics, slot transfers, and replay checks. The host executes a declared tool only after the artifact selects the learned tool name and arguments.
|
| 189 |
+
|
| 190 |
+
Example `crm_tools.py`:
|
| 191 |
+
|
| 192 |
+
```python
|
| 193 |
+
def lookup_customer(customer_id: object) -> dict[str, object]:
|
| 194 |
+
return {
|
| 195 |
+
"customer_id": str(customer_id),
|
| 196 |
+
"tier": "enterprise",
|
| 197 |
+
"renewal_days": 19,
|
| 198 |
+
}
|
| 199 |
+
```
|
| 200 |
+
|
| 201 |
+
Example `reasoning_evidence.py`:
|
| 202 |
+
|
| 203 |
+
```python
|
| 204 |
+
from __future__ import annotations
|
| 205 |
+
|
| 206 |
+
import json
|
| 207 |
+
import subprocess
|
| 208 |
+
import sys
|
| 209 |
+
from pathlib import Path
|
| 210 |
+
|
| 211 |
+
RUNTIME = Path(__file__).with_name('sle_spark.py')
|
| 212 |
+
ARTIFACT = Path(__file__).with_name("model.cra")
|
| 213 |
+
|
| 214 |
+
|
| 215 |
+
def spark_command() -> list[str]:
|
| 216 |
+
if RUNTIME.suffix.lower() in {".py", ".pyz"}:
|
| 217 |
+
return [sys.executable, str(RUNTIME)]
|
| 218 |
+
return [str(RUNTIME)]
|
| 219 |
+
|
| 220 |
+
|
| 221 |
+
def run_saccadic(mode: str, text: str, *extra: str) -> dict[str, object]:
|
| 222 |
+
completed = subprocess.run(
|
| 223 |
+
spark_command() + [
|
| 224 |
+
"run-artifact",
|
| 225 |
+
"--artifact", str(ARTIFACT),
|
| 226 |
+
"--mode", mode,
|
| 227 |
+
"--text", text,
|
| 228 |
+
*extra,
|
| 229 |
+
],
|
| 230 |
+
check=False,
|
| 231 |
+
capture_output=True,
|
| 232 |
+
text=True,
|
| 233 |
+
encoding="utf-8",
|
| 234 |
+
)
|
| 235 |
+
output = completed.stdout or completed.stderr
|
| 236 |
+
payload = json.loads(output)
|
| 237 |
+
payload["returncode"] = completed.returncode
|
| 238 |
+
return payload
|
| 239 |
+
|
| 240 |
+
|
| 241 |
+
payload = run_saccadic(
|
| 242 |
+
"autonomous",
|
| 243 |
+
"Look up customer CUST-42 and summarize renewal state.",
|
| 244 |
+
"--host-tool", "crm.lookup_customer=crm_tools:lookup_customer",
|
| 245 |
+
"--avoid-replay",
|
| 246 |
+
)
|
| 247 |
+
|
| 248 |
+
reasoning_keys = (
|
| 249 |
+
"success",
|
| 250 |
+
"returncode",
|
| 251 |
+
"error",
|
| 252 |
+
"route",
|
| 253 |
+
"tool_path",
|
| 254 |
+
"arguments",
|
| 255 |
+
"final_value",
|
| 256 |
+
"slot_transfers",
|
| 257 |
+
"punctuation_symbols",
|
| 258 |
+
"stream",
|
| 259 |
+
"emitted_replay_evidence",
|
| 260 |
+
"blocked_replay_evidence",
|
| 261 |
+
)
|
| 262 |
+
for key in reasoning_keys:
|
| 263 |
+
print(key, json.dumps(payload.get(key), ensure_ascii=False))
|
| 264 |
+
```
|
| 265 |
+
|
| 266 |
+
### Conversational Chat
|
| 267 |
+
|
| 268 |
+
Keep recent turns in the host and send them through history mode. Saccadic selects the newest parseable suffix from the supplied transcript and returns the loaded-artifact reply plus evidence.
|
| 269 |
+
|
| 270 |
+
```python
|
| 271 |
+
from __future__ import annotations
|
| 272 |
+
|
| 273 |
+
import json
|
| 274 |
+
import subprocess
|
| 275 |
+
import sys
|
| 276 |
+
from pathlib import Path
|
| 277 |
+
|
| 278 |
+
RUNTIME = Path(__file__).with_name('sle_spark.py')
|
| 279 |
+
ARTIFACT = Path(__file__).with_name("model.cra")
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
def spark_command() -> list[str]:
|
| 283 |
+
if RUNTIME.suffix.lower() in {".py", ".pyz"}:
|
| 284 |
+
return [sys.executable, str(RUNTIME)]
|
| 285 |
+
return [str(RUNTIME)]
|
| 286 |
+
|
| 287 |
+
|
| 288 |
+
def run_saccadic(mode: str, text: str, *extra: str) -> dict[str, object]:
|
| 289 |
+
completed = subprocess.run(
|
| 290 |
+
spark_command() + [
|
| 291 |
+
"run-artifact",
|
| 292 |
+
"--artifact", str(ARTIFACT),
|
| 293 |
+
"--mode", mode,
|
| 294 |
+
"--text", text,
|
| 295 |
+
*extra,
|
| 296 |
+
],
|
| 297 |
+
check=False,
|
| 298 |
+
capture_output=True,
|
| 299 |
+
text=True,
|
| 300 |
+
encoding="utf-8",
|
| 301 |
+
)
|
| 302 |
+
output = completed.stdout or completed.stderr
|
| 303 |
+
payload = json.loads(output)
|
| 304 |
+
payload["returncode"] = completed.returncode
|
| 305 |
+
return payload
|
| 306 |
+
|
| 307 |
+
|
| 308 |
+
history: list[tuple[str, str]] = []
|
| 309 |
+
|
| 310 |
+
|
| 311 |
+
def ask(user_text: str) -> dict[str, object]:
|
| 312 |
+
history.append(("User", user_text))
|
| 313 |
+
transcript = " ".join(f"{role}: {text}" for role, text in history)
|
| 314 |
+
payload = run_saccadic("history", transcript, "--avoid-replay")
|
| 315 |
+
reply_text = str(payload.get("text", ""))
|
| 316 |
+
history.append(("Assistant", reply_text))
|
| 317 |
+
return payload
|
| 318 |
+
|
| 319 |
+
|
| 320 |
+
first = ask("Who created you?")
|
| 321 |
+
second = ask("Use the latest chat context and say what you remember.")
|
| 322 |
+
|
| 323 |
+
for payload in (first, second):
|
| 324 |
+
print(json.dumps({
|
| 325 |
+
"text": payload.get("text"),
|
| 326 |
+
"success": payload.get("success"),
|
| 327 |
+
"returncode": payload.get("returncode"),
|
| 328 |
+
"error": payload.get("error"),
|
| 329 |
+
"selected_segment_index": payload.get("selected_segment_index"),
|
| 330 |
+
"selected_suffix_start": payload.get("selected_suffix_start"),
|
| 331 |
+
"slot_transfers": payload.get("slot_transfers"),
|
| 332 |
+
"punctuation_symbols": payload.get("punctuation_symbols"),
|
| 333 |
+
"stream": payload.get("stream"),
|
| 334 |
+
"emitted_replay_evidence": payload.get("emitted_replay_evidence"),
|
| 335 |
+
}, ensure_ascii=False))
|
| 336 |
+
```
|
| 337 |
+
|
| 338 |
+
## Agentic Use
|
| 339 |
+
|
| 340 |
+
Bind host tools by learned artifact tool name. Saccadic selects the tool path, emits the argument map, consumes the returned host value, and generates the reply through artifact state.
|
| 341 |
+
|
| 342 |
+
Example `crm_tools.py`:
|
| 343 |
+
|
| 344 |
+
```python
|
| 345 |
+
def lookup_customer(customer_id: object) -> dict[str, object]:
|
| 346 |
+
record = crm_client.lookup_customer(str(customer_id))
|
| 347 |
+
return {
|
| 348 |
+
"customer_id": record.id,
|
| 349 |
+
"tier": record.tier,
|
| 350 |
+
"renewal_days": record.renewal_days,
|
| 351 |
+
}
|
| 352 |
+
```
|
| 353 |
+
|
| 354 |
+
Run a custom tool-bound turn:
|
| 355 |
+
|
| 356 |
+
```powershell
|
| 357 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode autonomous --text "Look up customer CUST-42 and summarize renewal state." --host-tool "crm.lookup_customer=crm_tools:lookup_customer" --avoid-replay
|
| 358 |
+
```
|
| 359 |
+
|
| 360 |
+
For repeated tool-backed action turns, inspect `autonomous_action_unique_text_count`, `autonomous_action_distinct_punctuation_symbols`, `autonomous_action_tool_paths`, `autonomous_action_arguments`, `autonomous_action_final_values`, `autonomous_action_slot_transfers_by_turn`, `autonomous_action_transition_use_counts_before_turn`, `autonomous_action_transition_use_counts_after_turn`, emitted exact-row replay evidence, and blocked replay evidence:
|
| 361 |
+
|
| 362 |
+
```powershell
|
| 363 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode autonomous --text "Look up customer CUST-42 and summarize renewal state." --turns 3 --diverse-speech-limit 3 --host-tool "crm.lookup_customer=crm_tools:lookup_customer" --output-artifact saccadic_after_actions.cra --avoid-replay
|
| 364 |
+
```
|
| 365 |
+
|
| 366 |
+
### HTTP Service
|
| 367 |
+
|
| 368 |
+
Start the HTTP boundary:
|
| 369 |
+
|
| 370 |
+
```powershell
|
| 371 |
+
python .\sle_spark.py serve-artifact --artifact model.cra --host 0.0.0.0 --port 8000
|
| 372 |
+
```
|
| 373 |
+
|
| 374 |
+
Send JSON to `POST /run`:
|
| 375 |
+
|
| 376 |
+
```json
|
| 377 |
+
{
|
| 378 |
+
"mode": "response",
|
| 379 |
+
"text": "System: Reply directly without visible thoughts. User: Name one calm color."
|
| 380 |
+
}
|
| 381 |
+
```
|
| 382 |
+
|
| 383 |
+
The service returns selected instructions, selected tool paths, emitted arguments, final values, Speech Cortex slot transfers, stream dynamics, and replay evidence.
|
| 384 |
+
|
| 385 |
+
### OpenAI-Compatible Chat API
|
| 386 |
+
|
| 387 |
+
The same service exposes `POST /v1/chat/completions` for OpenAI SDK clients. Start the artifact service, then point the SDK at the local Spark boundary:
|
| 388 |
+
|
| 389 |
+
```powershell
|
| 390 |
+
python .\sle_spark.py serve-artifact --artifact model.cra --host 127.0.0.1 --port 8765
|
| 391 |
+
python -m pip install openai
|
| 392 |
+
```
|
| 393 |
+
|
| 394 |
+
Example `chat_app.py`:
|
| 395 |
+
|
| 396 |
+
```python
|
| 397 |
+
from __future__ import annotations
|
| 398 |
+
|
| 399 |
+
import json
|
| 400 |
+
from openai import OpenAI
|
| 401 |
+
|
| 402 |
+
|
| 403 |
+
client = OpenAI(
|
| 404 |
+
base_url="http://127.0.0.1:8765/v1",
|
| 405 |
+
api_key="not-needed",
|
| 406 |
+
)
|
| 407 |
+
|
| 408 |
+
messages = [
|
| 409 |
+
{"role": "system", "content": "Reply naturally and use the latest user turn."},
|
| 410 |
+
]
|
| 411 |
+
|
| 412 |
+
|
| 413 |
+
def ask(user_text: str) -> str:
|
| 414 |
+
messages.append({"role": "user", "content": user_text})
|
| 415 |
+
completion = client.chat.completions.create(
|
| 416 |
+
model='SLE-V2.1-Ω12K-LX',
|
| 417 |
+
messages=messages,
|
| 418 |
+
extra_body={
|
| 419 |
+
"max_speech_steps": 18,
|
| 420 |
+
"avoid_texts": [],
|
| 421 |
+
},
|
| 422 |
+
)
|
| 423 |
+
raw = completion.model_dump()
|
| 424 |
+
reply = completion.choices[0].message.content or ""
|
| 425 |
+
messages.append({"role": "assistant", "content": reply})
|
| 426 |
+
print(json.dumps({
|
| 427 |
+
"text": reply,
|
| 428 |
+
"selected_symbols": raw["saccadic"].get("selected_symbols"),
|
| 429 |
+
"selected_suffix_start": raw["saccadic"].get("selected_suffix_start"),
|
| 430 |
+
"tool_path": raw["saccadic"].get("tool_path"),
|
| 431 |
+
"arguments": raw["saccadic"].get("arguments"),
|
| 432 |
+
"final_value": raw["saccadic"].get("final_value"),
|
| 433 |
+
"slot_transfers": raw["saccadic"].get("slot_transfers"),
|
| 434 |
+
"replay": raw["saccadic"].get("emitted_replay_evidence"),
|
| 435 |
+
}, ensure_ascii=False))
|
| 436 |
+
return reply
|
| 437 |
+
|
| 438 |
+
|
| 439 |
+
ask("Who created you?")
|
| 440 |
+
ask("Use the latest chat context and say what you remember.")
|
| 441 |
+
```
|
| 442 |
+
|
| 443 |
+
Tool-bound SDK request:
|
| 444 |
+
|
| 445 |
+
```python
|
| 446 |
+
tool_completion = client.chat.completions.create(
|
| 447 |
+
model='SLE-V2.1-Ω12K-LX',
|
| 448 |
+
messages=[
|
| 449 |
+
{"role": "system", "content": "Use the declared host tool when the artifact selects it."},
|
| 450 |
+
{"role": "user", "content": "Look up customer CUST-42 and summarize renewal state."},
|
| 451 |
+
],
|
| 452 |
+
extra_body={
|
| 453 |
+
"host_tools": ["crm.lookup_customer=crm_tools:lookup_customer"],
|
| 454 |
+
"max_speech_steps": 18,
|
| 455 |
+
},
|
| 456 |
+
)
|
| 457 |
+
tool_raw = tool_completion.model_dump()
|
| 458 |
+
print(json.dumps({
|
| 459 |
+
"text": tool_completion.choices[0].message.content,
|
| 460 |
+
"tool_path": tool_raw["saccadic"].get("tool_path"),
|
| 461 |
+
"arguments": tool_raw["saccadic"].get("arguments"),
|
| 462 |
+
"final_value": tool_raw["saccadic"].get("final_value"),
|
| 463 |
+
"slot_transfers": tool_raw["saccadic"].get("slot_transfers"),
|
| 464 |
+
"replay": tool_raw["saccadic"].get("emitted_replay_evidence"),
|
| 465 |
+
}, ensure_ascii=False))
|
| 466 |
+
```
|
| 467 |
+
|
| 468 |
+
Streaming response:
|
| 469 |
+
|
| 470 |
+
```python
|
| 471 |
+
stream = client.chat.completions.create(
|
| 472 |
+
model='SLE-V2.1-Ω12K-LX',
|
| 473 |
+
messages=messages + [{"role": "user", "content": "Continue from the latest chat context."}],
|
| 474 |
+
stream=True,
|
| 475 |
+
extra_body={
|
| 476 |
+
"max_speech_steps": 18,
|
| 477 |
+
},
|
| 478 |
+
)
|
| 479 |
+
|
| 480 |
+
evidence = None
|
| 481 |
+
for event in stream:
|
| 482 |
+
delta = event.choices[0].delta.content or ""
|
| 483 |
+
print(delta, end="")
|
| 484 |
+
extra = getattr(event, "model_extra", {}) or {}
|
| 485 |
+
if "saccadic" in extra:
|
| 486 |
+
evidence = extra["saccadic"]
|
| 487 |
+
print()
|
| 488 |
+
print(json.dumps(evidence, ensure_ascii=False))
|
| 489 |
+
```
|
| 490 |
+
|
| 491 |
+
### Edge And Desktop
|
| 492 |
+
|
| 493 |
+
Use the generated host files when included:
|
| 494 |
+
|
| 495 |
+
```powershell
|
| 496 |
+
python edge_host.py
|
| 497 |
+
python desktop_host.py
|
| 498 |
+
```
|
| 499 |
+
|
| 500 |
+
Both read JSON payloads from stdin, forward declared host tools to the shipped runtime, and print loaded-artifact evidence.
|
| 501 |
+
|
| 502 |
+
### Raw Acoustic Input
|
| 503 |
+
|
| 504 |
+
Raw acoustic runs pass waveform samples directly; there is no speech-to-text or tokenizer layer:
|
| 505 |
+
|
| 506 |
+
```powershell
|
| 507 |
+
python .\sle_spark.py run-artifact --artifact model.cra --mode acoustic --audio-samples "[0.0,0.17,0.34,0.17,0.0,-0.13,-0.30,-0.13,0.0]" --audio-symbol-score-floor 0.25 --diverse-speech-limit 2
|
| 508 |
+
```
|
| 509 |
+
|
| 510 |
+
## Best Practices
|
| 511 |
+
|
| 512 |
+
- Keep `model.cra`, Spark, `requirements.txt`, `SLE_RELEASE.json`, and any metrics files in the same deployment directory.
|
| 513 |
+
- Use `--avoid-replay` for public demos so exact source-row echoes are surfaced as evidence instead of mistaken for intelligence.
|
| 514 |
+
- Bind host tools with `--host-tool artifact.name=module:function`; the host executes tools, while Saccadic selects paths and emits arguments from loaded artifact state.
|
| 515 |
+
- Spark includes documented portable primitives for `math.add` and `math.power`; external tools still use explicit host bindings.
|
| 516 |
+
- Pass recent conversation as raw text in history mode when you want Saccadic to respond to prior turns.
|
| 517 |
+
- Inspect returned JSON fields such as selected tool paths, emitted arguments, final values, stream dynamics, punctuation symbols, slot transfers, and replay evidence.
|
| 518 |
+
|
| 519 |
+
## Trust And Evidence
|
| 520 |
+
|
| 521 |
+
Saccadic is an artifact-first release: supported response, history, autonomous, action, acoustic, and system-instruction paths return loaded-state evidence rather than hidden host-written answers. Evidence includes tool paths, emitted arguments, final values, autonomous route records, context transfers, punctuation symbols, Speech Cortex usage counts, system-instruction records, fact attributes, slot transfers, emitted exact-row replay evidence, and blocked replay evidence.
|
| 522 |
+
|
| 523 |
+
## Training And Selection
|
| 524 |
+
|
| 525 |
+
- Selected rows: 502,000
|
| 526 |
+
- Curation failures: 0
|
| 527 |
+
|
| 528 |
+
|
| 529 |
+
### Training Domains
|
| 530 |
+
|
| 531 |
+
- arts: 20,080
|
| 532 |
+
- biology: 20,080
|
| 533 |
+
- chemistry: 20,080
|
| 534 |
+
- constitution: 20,080
|
| 535 |
+
- conversation: 20,080
|
| 536 |
+
- economics: 20,080
|
| 537 |
+
- emoji: 20,080
|
| 538 |
+
- geography: 20,080
|
| 539 |
+
- history: 20,080
|
| 540 |
+
- language_es: 20,080
|
| 541 |
+
- language_fr: 20,080
|
| 542 |
+
- language_ha: 20,080
|
| 543 |
+
- language_ig: 20,080
|
| 544 |
+
- language_yo: 20,080
|
| 545 |
+
- language_zh: 20,080
|
| 546 |
+
- law: 20,080
|
| 547 |
+
- math: 20,080
|
| 548 |
+
- medicine: 20,080
|
| 549 |
+
- philosophy: 20,080
|
| 550 |
+
- physics: 20,080
|
| 551 |
+
- safety: 20,080
|
| 552 |
+
- stories: 20,080
|
| 553 |
+
- system_instruction: 20,080
|
| 554 |
+
- technology: 20,080
|
| 555 |
+
- world: 20,080
|
| 556 |
+
|
| 557 |
+
## Citation
|
| 558 |
+
|
| 559 |
+
```bibtex
|
| 560 |
+
@software{sle_saccadic,
|
| 561 |
+
title = {Saccadic-Liquid Engine: Saccadic Cognitive Resonance Artifact},
|
| 562 |
+
author = {OkeyMeta Ltd},
|
| 563 |
+
version = {SLE-V2.1-Ω12K-LX},
|
| 564 |
+
note = {CPU-first Cognitive Resonance Artifact with Spark runtime}
|
| 565 |
+
}
|
| 566 |
+
```
|
SLE_RELEASE.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"dataset":{"domain_counts":{"arts":20080,"biology":20080,"chemistry":20080,"constitution":20080,"conversation":20080,"economics":20080,"emoji":20080,"geography":20080,"history":20080,"language_es":20080,"language_fr":20080,"language_ha":20080,"language_ig":20080,"language_yo":20080,"language_zh":20080,"law":20080,"math":20080,"medicine":20080,"philosophy":20080,"physics":20080,"safety":20080,"stories":20080,"system_instruction":20080,"technology":20080,"world":20080},"failure_count":0,"row_count":502000},"deployment_entrypoints":{"cli":"./sle_spark.py run-artifact --artifact model.cra","container":"Containerfile","desktop":"python desktop_host.py","edge":"python edge_host.py","service":"./sle_spark.py serve-artifact --artifact model.cra --host 0.0.0.0 --port 8000"},"deployment_files":{"container":"Containerfile","desktop":"desktop_host.py","edge":"edge_host.py"},"deployment_targets":["cli","service","edge","desktop","container"],"files":{"artifact":"model.cra","metrics":"sle_v2_1_release_metrics.json","model_card":"README.md","requirements":"requirements.txt","spark_runtime":"sle_spark.py"},"hf_repo_slug":"SLE-V2.1-Omega12K-LX","internal_artifact_name":"saccadic_v2_1_12k_dialogue.cra","model_name":"Saccadic","omega_bindings":12000,"public_artifact_name":"model.cra","release_name":"SLE-V2.1-Ω12K-LX"}
|
desktop_host.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SLE desktop host boundary.
|
| 2 |
+
# This launcher calls the shipped Saccadic runtime against model.cra and prints artifact evidence.
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import json
|
| 6 |
+
import subprocess
|
| 7 |
+
import sys
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
|
| 10 |
+
RUNTIME_NAME = "sle_spark.py"
|
| 11 |
+
|
| 12 |
+
def _sequence(value):
|
| 13 |
+
if value is None:
|
| 14 |
+
return ()
|
| 15 |
+
if isinstance(value, str):
|
| 16 |
+
return (value,)
|
| 17 |
+
return tuple(value)
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def _append_optional(args, payload, key, flag):
|
| 21 |
+
value = payload.get(key)
|
| 22 |
+
if value is not None:
|
| 23 |
+
args.extend((flag, str(value)))
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def _run_artifact_args(payload):
|
| 27 |
+
mode = str(payload.get('mode', 'response'))
|
| 28 |
+
args = ['run-artifact', '--artifact', 'model.cra', '--mode', mode]
|
| 29 |
+
if mode == 'acoustic':
|
| 30 |
+
args.extend(('--audio-samples', json.dumps(payload.get('audio_samples', []))))
|
| 31 |
+
else:
|
| 32 |
+
args.extend(('--text', str(payload.get('text', ''))))
|
| 33 |
+
for tool in _sequence(payload.get('host_tools', payload.get('host_tool'))):
|
| 34 |
+
args.extend(('--host-tool', str(tool)))
|
| 35 |
+
if payload.get('avoid_replay'):
|
| 36 |
+
args.append('--avoid-replay')
|
| 37 |
+
for text in _sequence(payload.get('avoid_texts')):
|
| 38 |
+
args.extend(('--avoid-text', str(text)))
|
| 39 |
+
if payload.get('compose_speech'):
|
| 40 |
+
args.append('--compose-speech')
|
| 41 |
+
if payload.get('prefer_composed'):
|
| 42 |
+
args.append('--prefer-composed')
|
| 43 |
+
if payload.get('canonicalize_stream'):
|
| 44 |
+
args.append('--canonicalize-stream')
|
| 45 |
+
if payload.get('recover_on_failure'):
|
| 46 |
+
args.append('--recover-on-failure')
|
| 47 |
+
_append_optional(args, payload, 'audio_symbol_score_floor', '--audio-symbol-score-floor')
|
| 48 |
+
_append_optional(args, payload, 'diverse_speech_limit', '--diverse-speech-limit')
|
| 49 |
+
_append_optional(args, payload, 'turns', '--turns')
|
| 50 |
+
_append_optional(args, payload, 'punctuation_floor', '--punctuation-floor')
|
| 51 |
+
_append_optional(args, payload, 'output_artifact', '--output-artifact')
|
| 52 |
+
_append_optional(args, payload, 'dt', '--dt')
|
| 53 |
+
_append_optional(args, payload, 'max_speech_steps', '--max-speech-steps')
|
| 54 |
+
return args
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def _runtime_command(args):
|
| 58 |
+
runtime = Path(__file__).resolve().with_name(RUNTIME_NAME)
|
| 59 |
+
if runtime.is_file():
|
| 60 |
+
if runtime.suffix.lower() == '.py':
|
| 61 |
+
return [sys.executable, str(runtime), *args]
|
| 62 |
+
return [str(runtime), *args]
|
| 63 |
+
return [sys.executable, '-m', 'sle.cli', *args]
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def main() -> int:
|
| 67 |
+
payload = json.loads(sys.stdin.read() or '{}')
|
| 68 |
+
completed = subprocess.run(
|
| 69 |
+
_runtime_command(_run_artifact_args(payload)),
|
| 70 |
+
check=False,
|
| 71 |
+
text=True,
|
| 72 |
+
capture_output=True,
|
| 73 |
+
)
|
| 74 |
+
sys.stdout.write(completed.stdout)
|
| 75 |
+
if completed.stderr:
|
| 76 |
+
sys.stderr.write(completed.stderr)
|
| 77 |
+
return int(completed.returncode)
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
if __name__ == '__main__':
|
| 81 |
+
raise SystemExit(main())
|
edge_host.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SLE edge host boundary.
|
| 2 |
+
# This file forwards payloads to the shipped Saccadic runtime; intelligence stays in model.cra.
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import json
|
| 6 |
+
import subprocess
|
| 7 |
+
import sys
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
|
| 10 |
+
RUNTIME_NAME = "sle_spark.py"
|
| 11 |
+
|
| 12 |
+
def _sequence(value):
|
| 13 |
+
if value is None:
|
| 14 |
+
return ()
|
| 15 |
+
if isinstance(value, str):
|
| 16 |
+
return (value,)
|
| 17 |
+
return tuple(value)
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def _append_optional(args, payload, key, flag):
|
| 21 |
+
value = payload.get(key)
|
| 22 |
+
if value is not None:
|
| 23 |
+
args.extend((flag, str(value)))
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def _run_artifact_args(payload):
|
| 27 |
+
mode = str(payload.get('mode', 'response'))
|
| 28 |
+
args = ['run-artifact', '--artifact', 'model.cra', '--mode', mode]
|
| 29 |
+
if mode == 'acoustic':
|
| 30 |
+
args.extend(('--audio-samples', json.dumps(payload.get('audio_samples', []))))
|
| 31 |
+
else:
|
| 32 |
+
args.extend(('--text', str(payload.get('text', ''))))
|
| 33 |
+
for tool in _sequence(payload.get('host_tools', payload.get('host_tool'))):
|
| 34 |
+
args.extend(('--host-tool', str(tool)))
|
| 35 |
+
if payload.get('avoid_replay'):
|
| 36 |
+
args.append('--avoid-replay')
|
| 37 |
+
for text in _sequence(payload.get('avoid_texts')):
|
| 38 |
+
args.extend(('--avoid-text', str(text)))
|
| 39 |
+
if payload.get('compose_speech'):
|
| 40 |
+
args.append('--compose-speech')
|
| 41 |
+
if payload.get('prefer_composed'):
|
| 42 |
+
args.append('--prefer-composed')
|
| 43 |
+
if payload.get('canonicalize_stream'):
|
| 44 |
+
args.append('--canonicalize-stream')
|
| 45 |
+
if payload.get('recover_on_failure'):
|
| 46 |
+
args.append('--recover-on-failure')
|
| 47 |
+
_append_optional(args, payload, 'audio_symbol_score_floor', '--audio-symbol-score-floor')
|
| 48 |
+
_append_optional(args, payload, 'diverse_speech_limit', '--diverse-speech-limit')
|
| 49 |
+
_append_optional(args, payload, 'turns', '--turns')
|
| 50 |
+
_append_optional(args, payload, 'punctuation_floor', '--punctuation-floor')
|
| 51 |
+
_append_optional(args, payload, 'output_artifact', '--output-artifact')
|
| 52 |
+
_append_optional(args, payload, 'dt', '--dt')
|
| 53 |
+
_append_optional(args, payload, 'max_speech_steps', '--max-speech-steps')
|
| 54 |
+
return args
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def _runtime_command(args):
|
| 58 |
+
runtime = Path(__file__).resolve().with_name(RUNTIME_NAME)
|
| 59 |
+
if runtime.is_file():
|
| 60 |
+
if runtime.suffix.lower() == '.py':
|
| 61 |
+
return [sys.executable, str(runtime), *args]
|
| 62 |
+
return [str(runtime), *args]
|
| 63 |
+
return [sys.executable, '-m', 'sle.cli', *args]
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def main() -> int:
|
| 67 |
+
payload = json.loads(sys.stdin.read() or '{}')
|
| 68 |
+
completed = subprocess.run(
|
| 69 |
+
_runtime_command(_run_artifact_args(payload)),
|
| 70 |
+
check=False,
|
| 71 |
+
text=True,
|
| 72 |
+
capture_output=True,
|
| 73 |
+
)
|
| 74 |
+
sys.stdout.write(completed.stdout)
|
| 75 |
+
if completed.stderr:
|
| 76 |
+
sys.stderr.write(completed.stderr)
|
| 77 |
+
return int(completed.returncode)
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
if __name__ == '__main__':
|
| 81 |
+
raise SystemExit(main())
|
model.cra
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1ee31736e1d1b051da62de5ab83dad73bab74dbd95b1196733d79c345bbe21ea
|
| 3 |
+
size 173909152
|
requirements.txt
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Public runtime reinstall contract for Saccadic.
|
| 2 |
+
# No private repository checkout is required to run model.cra with the shipped runtime.
|
| 3 |
+
setuptools>=68
|
| 4 |
+
numba>=0.65.1
|
| 5 |
+
llvmlite>=0.47.0
|
| 6 |
+
datasets>=4.1.1
|
sle_spark.py
ADDED
|
@@ -0,0 +1,1976 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import importlib
|
| 5 |
+
import json
|
| 6 |
+
import struct
|
| 7 |
+
import sys
|
| 8 |
+
import tempfile
|
| 9 |
+
import time
|
| 10 |
+
from http.server import BaseHTTPRequestHandler, HTTPServer
|
| 11 |
+
from pathlib import Path
|
| 12 |
+
from typing import Any
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
MAGIC = b"SLECRA1\0"
|
| 16 |
+
TOOL_MAGIC = b"SLETOOL1"
|
| 17 |
+
COGNITION_MAGIC = b"SLECOG1\0"
|
| 18 |
+
WORLD_MAGIC = b"SLEWRLD1"
|
| 19 |
+
LANGUAGE_MAGIC = b"SLELANG1"
|
| 20 |
+
DIALOGUE_MAGIC = b"SLEDIAL1"
|
| 21 |
+
INITIATIVE_MAGIC = b"SLEINIT1"
|
| 22 |
+
AUTONOMOUS_GOAL_MAGIC = b"SLEGOAL1"
|
| 23 |
+
AUTONOMOUS_ROUTE_MAGIC = b"SLEROUT1"
|
| 24 |
+
FACT_MAGIC = b"SLEFACT1"
|
| 25 |
+
STREAM_MAGIC = b"SLESTRM1"
|
| 26 |
+
SPEECH_MAGIC = b"SLESPEK1"
|
| 27 |
+
ACOUSTIC_MAGIC = b"SLEAUDI1"
|
| 28 |
+
ACOUSTIC_SYMBOL_MAGIC = b"SLEASMB1"
|
| 29 |
+
|
| 30 |
+
PUNCTUATION = frozenset((".", ",", "!", "?", ";", ":", "<", ">", "\u061f", "\u3002", "\uff01", "\uff1f"))
|
| 31 |
+
TERMINAL_PUNCTUATION = frozenset((".", "!", "?", "\u061f", "\u3002", "\uff01", "\uff1f"))
|
| 32 |
+
ROLE_LABELS = frozenset(("system", "developer", "user", "assistant", "tool"))
|
| 33 |
+
MASK_64 = 0xFFFFFFFFFFFFFFFF
|
| 34 |
+
FNV_OFFSET = 0xCBF29CE484222325
|
| 35 |
+
FNV_PRIME = 0x100000001B3
|
| 36 |
+
ACOUSTIC_SIGNATURE_NAMES = (
|
| 37 |
+
"length",
|
| 38 |
+
"first_bin",
|
| 39 |
+
"last_bin",
|
| 40 |
+
"mean_bin",
|
| 41 |
+
"energy_bin",
|
| 42 |
+
"peak_bin",
|
| 43 |
+
"crossings",
|
| 44 |
+
"direction",
|
| 45 |
+
)
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
def _read_exact(handle, length: int) -> bytes:
|
| 49 |
+
payload = handle.read(length)
|
| 50 |
+
if len(payload) != length:
|
| 51 |
+
raise ValueError("truncated CRA artifact")
|
| 52 |
+
return payload
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def _read_u32(handle) -> int:
|
| 56 |
+
return struct.unpack("<I", _read_exact(handle, 4))[0]
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def _read_u64(handle) -> int:
|
| 60 |
+
return struct.unpack("<Q", _read_exact(handle, 8))[0]
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def _read_block(handle) -> bytes:
|
| 64 |
+
return _read_exact(handle, _read_u32(handle))
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def _skip_block(handle) -> None:
|
| 68 |
+
_read_block(handle)
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def _read_json_block(handle) -> dict[str, Any]:
|
| 72 |
+
payload = _read_block(handle)
|
| 73 |
+
if not payload:
|
| 74 |
+
return {}
|
| 75 |
+
value = json.loads(payload.decode("utf-8"))
|
| 76 |
+
if not isinstance(value, dict):
|
| 77 |
+
raise ValueError("CRA JSON extension block must be an object")
|
| 78 |
+
return value
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def _load_artifact(path: str | Path) -> dict[str, Any]:
|
| 82 |
+
artifact: dict[str, Any] = {
|
| 83 |
+
"metadata": {},
|
| 84 |
+
"tools": {},
|
| 85 |
+
"cognition": {},
|
| 86 |
+
"language": {},
|
| 87 |
+
"instructions": {},
|
| 88 |
+
"contexts": {},
|
| 89 |
+
"response_policy": {},
|
| 90 |
+
"initiative": {},
|
| 91 |
+
"autonomous_goals": {},
|
| 92 |
+
"autonomous_routes": {},
|
| 93 |
+
"facts": {},
|
| 94 |
+
"stream": {},
|
| 95 |
+
"speech": {},
|
| 96 |
+
}
|
| 97 |
+
with Path(path).open("rb") as handle:
|
| 98 |
+
if _read_exact(handle, len(MAGIC)) != MAGIC:
|
| 99 |
+
raise ValueError("not an SLE CRA artifact")
|
| 100 |
+
version = _read_u32(handle)
|
| 101 |
+
if version != 1:
|
| 102 |
+
raise ValueError(f"unsupported CRA version: {version}")
|
| 103 |
+
artifact["version"] = version
|
| 104 |
+
metadata_payload = _read_block(handle)
|
| 105 |
+
artifact["metadata"] = json.loads(metadata_payload.decode("utf-8")) if metadata_payload else {}
|
| 106 |
+
dimensions = _read_u32(handle)
|
| 107 |
+
artifact["memory_dimensions"] = dimensions
|
| 108 |
+
artifact["memory_seed"] = _read_u64(handle)
|
| 109 |
+
record_count = _read_u32(handle)
|
| 110 |
+
artifact["memory_records"] = record_count
|
| 111 |
+
for _ in range(record_count):
|
| 112 |
+
_skip_block(handle)
|
| 113 |
+
_skip_block(handle)
|
| 114 |
+
|
| 115 |
+
artifact["liquid_neurons"] = _read_u32(handle)
|
| 116 |
+
artifact["liquid_inputs"] = _read_u32(handle)
|
| 117 |
+
artifact["liquid_seed"] = _read_u64(handle)
|
| 118 |
+
for _ in range(5):
|
| 119 |
+
_skip_block(handle)
|
| 120 |
+
|
| 121 |
+
marker = handle.read(len(TOOL_MAGIC))
|
| 122 |
+
if not marker:
|
| 123 |
+
return artifact
|
| 124 |
+
if marker != TOOL_MAGIC:
|
| 125 |
+
raise ValueError("unknown CRA extension section")
|
| 126 |
+
tools: dict[str, Any] = {}
|
| 127 |
+
for _ in range(_read_u32(handle)):
|
| 128 |
+
name = _read_block(handle).decode("utf-8")
|
| 129 |
+
payload = json.loads(_read_block(handle).decode("utf-8"))
|
| 130 |
+
_skip_block(handle)
|
| 131 |
+
has_avoidance = bool(struct.unpack("<?", _read_exact(handle, 1))[0])
|
| 132 |
+
if has_avoidance:
|
| 133 |
+
_skip_block(handle)
|
| 134 |
+
tools[name] = {
|
| 135 |
+
"name": name,
|
| 136 |
+
"parameters": [str(item) for item in payload.get("parameters", [])],
|
| 137 |
+
"successes": int(payload.get("successes", 0)),
|
| 138 |
+
"failures": int(payload.get("failures", 0)),
|
| 139 |
+
}
|
| 140 |
+
artifact["tools"] = tools
|
| 141 |
+
|
| 142 |
+
marker = handle.read(len(COGNITION_MAGIC))
|
| 143 |
+
while marker:
|
| 144 |
+
if len(marker) != len(COGNITION_MAGIC):
|
| 145 |
+
raise ValueError("truncated CRA extension marker")
|
| 146 |
+
if marker == COGNITION_MAGIC:
|
| 147 |
+
artifact["cognition"] = _read_json_block(handle)
|
| 148 |
+
elif marker == WORLD_MAGIC:
|
| 149 |
+
_skip_block(handle)
|
| 150 |
+
elif marker == LANGUAGE_MAGIC:
|
| 151 |
+
payload = _read_json_block(handle)
|
| 152 |
+
artifact["language"] = payload.get("language") or {}
|
| 153 |
+
artifact["instructions"] = payload.get("instructions") or {}
|
| 154 |
+
artifact["contexts"] = payload.get("contexts") or {}
|
| 155 |
+
elif marker == DIALOGUE_MAGIC:
|
| 156 |
+
artifact["response_policy"] = _read_json_block(handle)
|
| 157 |
+
elif marker == INITIATIVE_MAGIC:
|
| 158 |
+
artifact["initiative"] = _read_json_block(handle)
|
| 159 |
+
elif marker == AUTONOMOUS_GOAL_MAGIC:
|
| 160 |
+
artifact["autonomous_goals"] = _read_json_block(handle)
|
| 161 |
+
elif marker == AUTONOMOUS_ROUTE_MAGIC:
|
| 162 |
+
artifact["autonomous_routes"] = _read_json_block(handle)
|
| 163 |
+
elif marker == FACT_MAGIC:
|
| 164 |
+
artifact["facts"] = _read_json_block(handle)
|
| 165 |
+
elif marker == STREAM_MAGIC:
|
| 166 |
+
artifact["stream"] = _read_json_block(handle)
|
| 167 |
+
elif marker == SPEECH_MAGIC:
|
| 168 |
+
artifact["speech"] = _read_json_block(handle)
|
| 169 |
+
elif marker == ACOUSTIC_MAGIC:
|
| 170 |
+
artifact["acoustic"] = _read_json_block(handle)
|
| 171 |
+
elif marker == ACOUSTIC_SYMBOL_MAGIC:
|
| 172 |
+
artifact["acoustic_symbols"] = _read_json_block(handle)
|
| 173 |
+
else:
|
| 174 |
+
raise ValueError("unknown CRA extension section")
|
| 175 |
+
marker = handle.read(len(COGNITION_MAGIC))
|
| 176 |
+
return artifact
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
def _is_punctuation(symbol: str) -> bool:
|
| 180 |
+
return len(symbol) == 1 and (symbol in PUNCTUATION or not symbol.isalnum())
|
| 181 |
+
|
| 182 |
+
|
| 183 |
+
def _symbol_class(symbol: str) -> str:
|
| 184 |
+
if len(symbol) == 1 and symbol.isdecimal():
|
| 185 |
+
return "decimal"
|
| 186 |
+
return "text"
|
| 187 |
+
|
| 188 |
+
|
| 189 |
+
def _segment_text(text: str) -> list[str]:
|
| 190 |
+
symbols: list[str] = []
|
| 191 |
+
current: list[str] = []
|
| 192 |
+
current_class = ""
|
| 193 |
+
|
| 194 |
+
def finish_current() -> None:
|
| 195 |
+
nonlocal current, current_class
|
| 196 |
+
if current:
|
| 197 |
+
symbols.append("".join(current))
|
| 198 |
+
current = []
|
| 199 |
+
current_class = ""
|
| 200 |
+
|
| 201 |
+
for character in text:
|
| 202 |
+
if character.isspace():
|
| 203 |
+
finish_current()
|
| 204 |
+
elif _is_punctuation(character):
|
| 205 |
+
finish_current()
|
| 206 |
+
symbols.append(character)
|
| 207 |
+
else:
|
| 208 |
+
character_class = _symbol_class(character)
|
| 209 |
+
if current and character_class != current_class:
|
| 210 |
+
finish_current()
|
| 211 |
+
current.append(character)
|
| 212 |
+
current_class = character_class
|
| 213 |
+
finish_current()
|
| 214 |
+
return symbols
|
| 215 |
+
|
| 216 |
+
|
| 217 |
+
def _render_text(symbols: list[str]) -> str:
|
| 218 |
+
text = ""
|
| 219 |
+
in_angle_tag = False
|
| 220 |
+
angle_tag_content = ""
|
| 221 |
+
for symbol in symbols:
|
| 222 |
+
if symbol == "<":
|
| 223 |
+
text = text.rstrip() + "<"
|
| 224 |
+
in_angle_tag = True
|
| 225 |
+
angle_tag_content = ""
|
| 226 |
+
elif in_angle_tag:
|
| 227 |
+
if symbol == ">":
|
| 228 |
+
text = text.rstrip() + ">"
|
| 229 |
+
if angle_tag_content.startswith("/"):
|
| 230 |
+
text += " "
|
| 231 |
+
in_angle_tag = False
|
| 232 |
+
else:
|
| 233 |
+
text += symbol
|
| 234 |
+
angle_tag_content += symbol
|
| 235 |
+
elif _is_punctuation(symbol):
|
| 236 |
+
text = text.rstrip() + symbol + " "
|
| 237 |
+
else:
|
| 238 |
+
text += symbol + " "
|
| 239 |
+
return text.rstrip()
|
| 240 |
+
|
| 241 |
+
|
| 242 |
+
def _render_slot_value(symbols: list[str], example: Any) -> str:
|
| 243 |
+
example_text = str(example).strip()
|
| 244 |
+
if example_text and " " not in example_text:
|
| 245 |
+
return "".join(symbols)
|
| 246 |
+
return _render_text(symbols)
|
| 247 |
+
|
| 248 |
+
|
| 249 |
+
def _edit_distance(left: str, right: str) -> int:
|
| 250 |
+
if left == right:
|
| 251 |
+
return 0
|
| 252 |
+
previous = list(range(len(right) + 1))
|
| 253 |
+
for index, left_ch in enumerate(left, start=1):
|
| 254 |
+
current = [index]
|
| 255 |
+
for other_index, right_ch in enumerate(right, start=1):
|
| 256 |
+
cost = 0 if left_ch == right_ch else 1
|
| 257 |
+
current.append(min(previous[other_index] + 1, current[-1] + 1, previous[other_index - 1] + cost))
|
| 258 |
+
previous = current
|
| 259 |
+
return previous[-1]
|
| 260 |
+
|
| 261 |
+
|
| 262 |
+
def _parse_audio_samples(raw: Any) -> list[float]:
|
| 263 |
+
if raw is None:
|
| 264 |
+
raise ValueError("acoustic mode requires --audio-samples")
|
| 265 |
+
if isinstance(raw, str):
|
| 266 |
+
text = raw.strip()
|
| 267 |
+
if not text:
|
| 268 |
+
raise ValueError("audio samples must not be empty")
|
| 269 |
+
if text.startswith("["):
|
| 270 |
+
payload = json.loads(text)
|
| 271 |
+
else:
|
| 272 |
+
payload = [part.strip() for part in text.split(",") if part.strip()]
|
| 273 |
+
else:
|
| 274 |
+
payload = raw
|
| 275 |
+
if not isinstance(payload, list):
|
| 276 |
+
raise ValueError("audio samples must be a JSON list or comma-separated numbers")
|
| 277 |
+
samples = [float(value) for value in payload]
|
| 278 |
+
if not samples:
|
| 279 |
+
raise ValueError("audio samples must not be empty")
|
| 280 |
+
return samples
|
| 281 |
+
|
| 282 |
+
|
| 283 |
+
def _stream_evidence(text: str, stream_payload: dict[str, Any], canonicalize: bool = False) -> dict[str, Any]:
|
| 284 |
+
raw_chunks = _segment_text(text)
|
| 285 |
+
chunks = list(raw_chunks)
|
| 286 |
+
alias_names = [""] * len(chunks)
|
| 287 |
+
if canonicalize:
|
| 288 |
+
for index, chunk in enumerate(list(chunks)):
|
| 289 |
+
for raw in stream_payload.get("aliases", []) or []:
|
| 290 |
+
observed = str(raw.get("observed", ""))
|
| 291 |
+
max_distance = int(raw.get("max_distance", 0))
|
| 292 |
+
if _edit_distance(chunk, observed) <= max_distance:
|
| 293 |
+
chunks[index] = str(raw.get("canonical", observed))
|
| 294 |
+
alias_names[index] = str(raw.get("name", ""))
|
| 295 |
+
break
|
| 296 |
+
|
| 297 |
+
learned_surfaces: set[str] = set()
|
| 298 |
+
for raw in stream_payload.get("chunk_surfaces", []) or []:
|
| 299 |
+
learned_surfaces.update(str(item) for item in raw.get("chunks", []) or [])
|
| 300 |
+
boundaries: list[str] = []
|
| 301 |
+
errors: list[float] = []
|
| 302 |
+
for index, chunk in enumerate(chunks):
|
| 303 |
+
if index == len(chunks) - 1:
|
| 304 |
+
boundary = "end"
|
| 305 |
+
elif _is_punctuation(chunk):
|
| 306 |
+
boundary = "punctuation"
|
| 307 |
+
elif index + 1 < len(chunks) and _symbol_class(chunk[-1:]) != _symbol_class(chunks[index + 1][:1]):
|
| 308 |
+
boundary = "category"
|
| 309 |
+
else:
|
| 310 |
+
boundary = "space"
|
| 311 |
+
boundaries.append(boundary)
|
| 312 |
+
if chunk in learned_surfaces or _is_punctuation(chunk):
|
| 313 |
+
errors.append(0.0)
|
| 314 |
+
elif learned_surfaces:
|
| 315 |
+
best = min(_edit_distance(chunk, learned) / max(len(chunk), len(learned), 1) for learned in learned_surfaces)
|
| 316 |
+
errors.append(float(best))
|
| 317 |
+
else:
|
| 318 |
+
errors.append(1.0 / max(len(chunk), 1))
|
| 319 |
+
mean_error = sum(errors) / len(errors) if errors else 0.0
|
| 320 |
+
return {
|
| 321 |
+
"chunks": chunks,
|
| 322 |
+
"raw_chunks": raw_chunks,
|
| 323 |
+
"alias_names": alias_names,
|
| 324 |
+
"chunk_boundaries": boundaries,
|
| 325 |
+
"chunk_mean_errors": errors,
|
| 326 |
+
"category_boundary_count": sum(1 for boundary in boundaries if boundary == "category"),
|
| 327 |
+
"mean_error": mean_error,
|
| 328 |
+
"boundary_impulses": len(boundaries),
|
| 329 |
+
"steps": len(text),
|
| 330 |
+
"algorithm": "standalone-predictive-semantic-chunking",
|
| 331 |
+
}
|
| 332 |
+
|
| 333 |
+
|
| 334 |
+
def _acoustic_stream_evidence(samples: list[float], acoustic_payload: dict[str, Any]) -> dict[str, Any]:
|
| 335 |
+
if not acoustic_payload:
|
| 336 |
+
raise ValueError("artifact does not contain acoustic stream state")
|
| 337 |
+
bins = int(acoustic_payload.get("bins", 0))
|
| 338 |
+
if bins < 2:
|
| 339 |
+
raise ValueError("acoustic bins must be at least 2")
|
| 340 |
+
transitions = [float(value) for value in acoustic_payload.get("transitions", []) or []]
|
| 341 |
+
if len(transitions) != bins * bins:
|
| 342 |
+
raise ValueError("acoustic transition table length mismatch")
|
| 343 |
+
error_floor = float(acoustic_payload.get("error_floor", 0.5))
|
| 344 |
+
silence_floor = float(acoustic_payload.get("silence_floor", 0.02))
|
| 345 |
+
errors = [0.0] * len(samples)
|
| 346 |
+
boundaries = [0.0] * len(samples)
|
| 347 |
+
denominator = max(bins - 1, 1)
|
| 348 |
+
for index in range(1, len(samples)):
|
| 349 |
+
previous = _acoustic_quantize(samples[index - 1], bins)
|
| 350 |
+
current = _acoustic_quantize(samples[index], bins)
|
| 351 |
+
transition = transitions[previous * bins + current]
|
| 352 |
+
error = 0.0
|
| 353 |
+
if transition <= 0.0:
|
| 354 |
+
error = abs(current - previous) / denominator
|
| 355 |
+
boundary = 0.0
|
| 356 |
+
if error >= error_floor:
|
| 357 |
+
boundary = 1.0
|
| 358 |
+
elif abs(samples[index]) <= silence_floor and abs(samples[index - 1]) > silence_floor:
|
| 359 |
+
boundary = 1.0
|
| 360 |
+
errors[index] = error
|
| 361 |
+
boundaries[index] = boundary
|
| 362 |
+
|
| 363 |
+
chunks = []
|
| 364 |
+
start = 0
|
| 365 |
+
for index in range(1, len(samples)):
|
| 366 |
+
if boundaries[index] <= 0.0:
|
| 367 |
+
continue
|
| 368 |
+
chunks.append(_acoustic_chunk(samples, errors, start, index, _acoustic_boundary_name(samples, errors, index, error_floor, silence_floor)))
|
| 369 |
+
start = index
|
| 370 |
+
chunks.append(_acoustic_chunk(samples, errors, start, len(samples), "end"))
|
| 371 |
+
chunks = [chunk for chunk in chunks if chunk["start"] < chunk["end"]]
|
| 372 |
+
return {
|
| 373 |
+
"chunks": chunks,
|
| 374 |
+
"chunk_spans": [[chunk["start"], chunk["end"]] for chunk in chunks],
|
| 375 |
+
"chunk_boundaries": [chunk["boundary"] for chunk in chunks],
|
| 376 |
+
"chunk_mean_errors": [chunk["mean_error"] for chunk in chunks],
|
| 377 |
+
"mean_error": sum(errors) / len(errors) if errors else 0.0,
|
| 378 |
+
"boundary_impulses": sum(1 for value in boundaries if value > 0.0),
|
| 379 |
+
"steps": len(samples),
|
| 380 |
+
"algorithm": "standalone-predictive-acoustic-chunking",
|
| 381 |
+
}
|
| 382 |
+
|
| 383 |
+
|
| 384 |
+
def _acoustic_chunk(samples: list[float], errors: list[float], start: int, end: int, boundary: str) -> dict[str, Any]:
|
| 385 |
+
width = max(end - start, 1)
|
| 386 |
+
mean_error = sum(errors[start:end]) / width if end > start else 0.0
|
| 387 |
+
return {"start": int(start), "end": int(end), "mean_error": float(mean_error), "boundary": boundary}
|
| 388 |
+
|
| 389 |
+
|
| 390 |
+
def _acoustic_boundary_name(samples: list[float], errors: list[float], index: int, error_floor: float, silence_floor: float) -> str:
|
| 391 |
+
if errors[index] >= error_floor:
|
| 392 |
+
return "prediction_error"
|
| 393 |
+
if index > 0 and abs(samples[index]) <= silence_floor and abs(samples[index - 1]) > silence_floor:
|
| 394 |
+
return "silence"
|
| 395 |
+
return "boundary"
|
| 396 |
+
|
| 397 |
+
|
| 398 |
+
def _acoustic_quantize(value: float, bins: int) -> int:
|
| 399 |
+
bounded = max(-1.0, min(1.0, float(value)))
|
| 400 |
+
scaled = (bounded + 1.0) * 0.5 * (bins - 1)
|
| 401 |
+
index = int(scaled + 0.5)
|
| 402 |
+
if index < 0:
|
| 403 |
+
return 0
|
| 404 |
+
if index >= bins:
|
| 405 |
+
return bins - 1
|
| 406 |
+
return index
|
| 407 |
+
|
| 408 |
+
|
| 409 |
+
def _decode_acoustic_symbols(
|
| 410 |
+
artifact: dict[str, Any],
|
| 411 |
+
samples: list[float],
|
| 412 |
+
chunks: list[dict[str, Any]],
|
| 413 |
+
min_score: float,
|
| 414 |
+
) -> dict[str, Any]:
|
| 415 |
+
payload = artifact.get("acoustic_symbols", {}) or {}
|
| 416 |
+
records = list(payload.get("records", []) or [])
|
| 417 |
+
if not records:
|
| 418 |
+
raise ValueError("artifact does not contain acoustic symbol grounding state")
|
| 419 |
+
bins = int(payload.get("bins", 0))
|
| 420 |
+
if bins < 2:
|
| 421 |
+
raise ValueError("acoustic symbol bins must be at least 2")
|
| 422 |
+
dimensions = int(artifact.get("memory_dimensions", 0))
|
| 423 |
+
seed = int(artifact.get("memory_seed", 0))
|
| 424 |
+
if dimensions <= 0:
|
| 425 |
+
raise ValueError("artifact memory dimensions must be positive")
|
| 426 |
+
|
| 427 |
+
matches = []
|
| 428 |
+
for chunk in chunks:
|
| 429 |
+
if _acoustic_chunk_is_silent(samples, chunk, 0.0):
|
| 430 |
+
continue
|
| 431 |
+
query = _hdc_encode(_acoustic_signature(samples, chunk["start"], chunk["end"], bins), dimensions, seed)
|
| 432 |
+
ranked = []
|
| 433 |
+
for raw in records:
|
| 434 |
+
encoded = _hdc_encode(dict(raw.get("signature", {}) or {}), dimensions, seed)
|
| 435 |
+
score = _vector_resonance(query, encoded)
|
| 436 |
+
ranked.append((score, str(raw.get("name", "")), raw))
|
| 437 |
+
ranked.sort(key=lambda item: (-item[0], item[1]))
|
| 438 |
+
score, _name, raw = ranked[0]
|
| 439 |
+
if score < min_score:
|
| 440 |
+
raise ValueError(f"acoustic symbol score below floor: {score}")
|
| 441 |
+
matches.append(
|
| 442 |
+
{
|
| 443 |
+
"symbol": str(raw.get("symbol", "")),
|
| 444 |
+
"record_name": str(raw.get("name", "")),
|
| 445 |
+
"score": float(score),
|
| 446 |
+
"chunk_span": [int(chunk["start"]), int(chunk["end"])],
|
| 447 |
+
"chunk_boundary": str(chunk["boundary"]),
|
| 448 |
+
}
|
| 449 |
+
)
|
| 450 |
+
return {
|
| 451 |
+
"symbols": [match["symbol"] for match in matches],
|
| 452 |
+
"matches": matches,
|
| 453 |
+
"algorithm": "standalone-hdc-acoustic-symbol-grounding",
|
| 454 |
+
}
|
| 455 |
+
|
| 456 |
+
|
| 457 |
+
def _acoustic_signature(samples: list[float], start: int, end: int, bins: int) -> dict[str, int]:
|
| 458 |
+
if end <= start:
|
| 459 |
+
return {name: 0 for name in ACOUSTIC_SIGNATURE_NAMES}
|
| 460 |
+
first = _acoustic_quantize(samples[start], bins)
|
| 461 |
+
last = _acoustic_quantize(samples[end - 1], bins)
|
| 462 |
+
total = 0.0
|
| 463 |
+
abs_total = 0.0
|
| 464 |
+
peak_abs = 0.0
|
| 465 |
+
crossings = 0
|
| 466 |
+
previous_sign = 0
|
| 467 |
+
for index in range(start, end):
|
| 468 |
+
value = float(samples[index])
|
| 469 |
+
total += value
|
| 470 |
+
abs_value = abs(value)
|
| 471 |
+
abs_total += abs_value
|
| 472 |
+
if abs_value > peak_abs:
|
| 473 |
+
peak_abs = abs_value
|
| 474 |
+
sign = 0
|
| 475 |
+
if value > 0.0:
|
| 476 |
+
sign = 1
|
| 477 |
+
elif value < 0.0:
|
| 478 |
+
sign = -1
|
| 479 |
+
if sign != 0:
|
| 480 |
+
if previous_sign != 0 and sign != previous_sign:
|
| 481 |
+
crossings += 1
|
| 482 |
+
previous_sign = sign
|
| 483 |
+
length = end - start
|
| 484 |
+
mean = total / length
|
| 485 |
+
energy = abs_total / length
|
| 486 |
+
energy_bin = _bounded_bin(energy * (bins - 1), bins)
|
| 487 |
+
peak_bin = _bounded_bin(peak_abs * (bins - 1), bins)
|
| 488 |
+
direction = 1
|
| 489 |
+
if samples[end - 1] > samples[start]:
|
| 490 |
+
direction = 2
|
| 491 |
+
elif samples[end - 1] < samples[start]:
|
| 492 |
+
direction = 0
|
| 493 |
+
return {
|
| 494 |
+
"length": int(length),
|
| 495 |
+
"first_bin": int(first),
|
| 496 |
+
"last_bin": int(last),
|
| 497 |
+
"mean_bin": int(_acoustic_quantize(mean, bins)),
|
| 498 |
+
"energy_bin": int(energy_bin),
|
| 499 |
+
"peak_bin": int(peak_bin),
|
| 500 |
+
"crossings": int(crossings),
|
| 501 |
+
"direction": int(direction),
|
| 502 |
+
}
|
| 503 |
+
|
| 504 |
+
|
| 505 |
+
def _bounded_bin(value: float, bins: int) -> int:
|
| 506 |
+
index = int(value + 0.5)
|
| 507 |
+
if index < 0:
|
| 508 |
+
return 0
|
| 509 |
+
if index >= bins:
|
| 510 |
+
return bins - 1
|
| 511 |
+
return index
|
| 512 |
+
|
| 513 |
+
|
| 514 |
+
def _acoustic_chunk_is_silent(samples: list[float], chunk: dict[str, Any], silence_floor: float) -> bool:
|
| 515 |
+
for index in range(int(chunk["start"]), int(chunk["end"])):
|
| 516 |
+
if abs(samples[index]) > silence_floor:
|
| 517 |
+
return False
|
| 518 |
+
return True
|
| 519 |
+
|
| 520 |
+
|
| 521 |
+
def _hdc_encode(attributes: dict[str, Any], dimensions: int, seed: int) -> list[float]:
|
| 522 |
+
if not attributes:
|
| 523 |
+
raise ValueError("HDC encode requires attributes")
|
| 524 |
+
accumulator = [0.0] * dimensions
|
| 525 |
+
for role in sorted(attributes):
|
| 526 |
+
role_vector = _hdc_symbol(str(role), dimensions, seed)
|
| 527 |
+
value_vector = _hdc_symbol(str(attributes[role]), dimensions, seed)
|
| 528 |
+
for index in range(dimensions):
|
| 529 |
+
accumulator[index] += role_vector[index] * value_vector[index]
|
| 530 |
+
out = [0.0] * dimensions
|
| 531 |
+
for index, value in enumerate(accumulator):
|
| 532 |
+
if value > 0.0:
|
| 533 |
+
out[index] = 1.0
|
| 534 |
+
elif value < 0.0:
|
| 535 |
+
out[index] = -1.0
|
| 536 |
+
elif index % 2 == 0:
|
| 537 |
+
out[index] = 1.0
|
| 538 |
+
else:
|
| 539 |
+
out[index] = -1.0
|
| 540 |
+
return out
|
| 541 |
+
|
| 542 |
+
|
| 543 |
+
def _hdc_symbol(name: str, dimensions: int, seed: int) -> list[float]:
|
| 544 |
+
base_hash = _stable_text_hash(name, seed)
|
| 545 |
+
seed_u64 = int(seed) & MASK_64
|
| 546 |
+
out = [0.0] * dimensions
|
| 547 |
+
for index in range(dimensions):
|
| 548 |
+
mixed = _mix64(base_hash ^ seed_u64 ^ index)
|
| 549 |
+
out[index] = 1.0 if mixed & 1 else -1.0
|
| 550 |
+
return out
|
| 551 |
+
|
| 552 |
+
|
| 553 |
+
def _stable_text_hash(text: str, seed: int) -> int:
|
| 554 |
+
value = (FNV_OFFSET ^ (int(seed) & MASK_64)) & MASK_64
|
| 555 |
+
for byte in text.encode("utf-8"):
|
| 556 |
+
value ^= byte
|
| 557 |
+
value = (value * FNV_PRIME) & MASK_64
|
| 558 |
+
return value
|
| 559 |
+
|
| 560 |
+
|
| 561 |
+
def _mix64(value: int) -> int:
|
| 562 |
+
z = (int(value) + 0x9E3779B97F4A7C15) & MASK_64
|
| 563 |
+
z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & MASK_64
|
| 564 |
+
z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & MASK_64
|
| 565 |
+
return (z ^ (z >> 31)) & MASK_64
|
| 566 |
+
|
| 567 |
+
|
| 568 |
+
def _vector_resonance(left: list[float], right: list[float]) -> float:
|
| 569 |
+
if len(left) != len(right) or not left:
|
| 570 |
+
raise ValueError("resonance vectors must have the same positive length")
|
| 571 |
+
total = 0.0
|
| 572 |
+
for left_value, right_value in zip(left, right):
|
| 573 |
+
total += left_value * right_value
|
| 574 |
+
return total / len(left)
|
| 575 |
+
|
| 576 |
+
|
| 577 |
+
def _match_pattern(pattern: list[str], symbols: list[str], examples: dict[str, Any]) -> tuple[dict[str, Any], float] | None:
|
| 578 |
+
slots: dict[str, Any] = {}
|
| 579 |
+
literal_scores: list[float] = []
|
| 580 |
+
pattern_index = 0
|
| 581 |
+
symbol_index = 0
|
| 582 |
+
while pattern_index < len(pattern):
|
| 583 |
+
if symbol_index >= len(symbols):
|
| 584 |
+
return None
|
| 585 |
+
expected = pattern[pattern_index]
|
| 586 |
+
if expected.startswith("$"):
|
| 587 |
+
key = expected[1:]
|
| 588 |
+
example = examples.get(key, "")
|
| 589 |
+
slot_width = max(1, len(_segment_text(str(example))))
|
| 590 |
+
if symbol_index + slot_width > len(symbols):
|
| 591 |
+
return None
|
| 592 |
+
observed_symbols = symbols[symbol_index:symbol_index + slot_width]
|
| 593 |
+
try:
|
| 594 |
+
slots[key] = _coerce_like(example, _render_slot_value(observed_symbols, example))
|
| 595 |
+
except (TypeError, ValueError):
|
| 596 |
+
return None
|
| 597 |
+
symbol_index += slot_width
|
| 598 |
+
else:
|
| 599 |
+
observed = symbols[symbol_index]
|
| 600 |
+
width = max(len(expected), len(observed), 1)
|
| 601 |
+
score = 1.0 - (_edit_distance(expected, observed) / width)
|
| 602 |
+
if score <= 0.0:
|
| 603 |
+
return None
|
| 604 |
+
literal_scores.append(score)
|
| 605 |
+
symbol_index += 1
|
| 606 |
+
pattern_index += 1
|
| 607 |
+
if symbol_index != len(symbols):
|
| 608 |
+
return None
|
| 609 |
+
score = sum(literal_scores) / len(literal_scores) if literal_scores else 1.0
|
| 610 |
+
return slots, score
|
| 611 |
+
|
| 612 |
+
|
| 613 |
+
def _goal_from_instruction(raw: dict[str, Any], slots: dict[str, Any]) -> dict[str, Any]:
|
| 614 |
+
goal = {}
|
| 615 |
+
for key, value in dict(raw.get("goal_template", {}) or {}).items():
|
| 616 |
+
if isinstance(value, str) and value.startswith("$"):
|
| 617 |
+
goal[str(key)] = slots[value[1:]]
|
| 618 |
+
else:
|
| 619 |
+
goal[str(key)] = value
|
| 620 |
+
return goal
|
| 621 |
+
|
| 622 |
+
|
| 623 |
+
def _coerce_like(example: Any, value: str) -> Any:
|
| 624 |
+
if isinstance(example, bool):
|
| 625 |
+
return value == "True"
|
| 626 |
+
if isinstance(example, int) and not isinstance(example, bool):
|
| 627 |
+
return int(value)
|
| 628 |
+
if isinstance(example, float):
|
| 629 |
+
return float(value)
|
| 630 |
+
return value
|
| 631 |
+
|
| 632 |
+
|
| 633 |
+
def _parse_instruction(artifact: dict[str, Any], symbols: list[str]) -> dict[str, Any]:
|
| 634 |
+
candidates: list[tuple[float, str, dict[str, Any]]] = []
|
| 635 |
+
for raw in artifact.get("instructions", {}).get("records", []) or []:
|
| 636 |
+
match = _match_pattern(
|
| 637 |
+
[str(item) for item in raw.get("pattern", [])],
|
| 638 |
+
symbols,
|
| 639 |
+
dict(raw.get("slot_examples", {}) or {}),
|
| 640 |
+
)
|
| 641 |
+
if match is None:
|
| 642 |
+
continue
|
| 643 |
+
slots, score = match
|
| 644 |
+
goal = _goal_from_instruction(raw, slots)
|
| 645 |
+
candidates.append((score, str(raw.get("name", "")), goal))
|
| 646 |
+
if not candidates:
|
| 647 |
+
raise ValueError("no instruction pattern matched")
|
| 648 |
+
candidates.sort(key=lambda item: (-item[0], item[1]))
|
| 649 |
+
score, name, goal = candidates[0]
|
| 650 |
+
return {"record_name": name, "goal": goal, "score": score}
|
| 651 |
+
|
| 652 |
+
|
| 653 |
+
def _system_context_priority(symbols: list[str], goal: dict[str, Any]) -> int:
|
| 654 |
+
if len(symbols) < 4:
|
| 655 |
+
return 0
|
| 656 |
+
has_system_role = symbols[0] == "System" and symbols[1] == ":"
|
| 657 |
+
has_user_role = any(
|
| 658 |
+
symbols[index] == "User" and symbols[index + 1] == ":"
|
| 659 |
+
for index in range(len(symbols) - 1)
|
| 660 |
+
)
|
| 661 |
+
has_system_goal = (
|
| 662 |
+
"system_prompt" in goal
|
| 663 |
+
or str(goal.get("query_kind", "")) == "system_instruction"
|
| 664 |
+
)
|
| 665 |
+
return 1 if has_system_role and has_user_role and has_system_goal else 0
|
| 666 |
+
|
| 667 |
+
|
| 668 |
+
def _parse_instruction_suffix(artifact: dict[str, Any], symbols: list[str]) -> dict[str, Any]:
|
| 669 |
+
candidates: list[tuple[float, int, int, int, str, dict[str, Any], list[str]]] = []
|
| 670 |
+
records = artifact.get("instructions", {}).get("records", []) or []
|
| 671 |
+
for suffix_start in range(len(symbols)):
|
| 672 |
+
candidate_symbols = symbols[suffix_start:]
|
| 673 |
+
if not candidate_symbols or _is_punctuation(candidate_symbols[0]):
|
| 674 |
+
continue
|
| 675 |
+
for raw in records:
|
| 676 |
+
pattern = [str(item) for item in raw.get("pattern", [])]
|
| 677 |
+
match = _match_pattern(
|
| 678 |
+
pattern,
|
| 679 |
+
candidate_symbols,
|
| 680 |
+
dict(raw.get("slot_examples", {}) or {}),
|
| 681 |
+
)
|
| 682 |
+
if match is None:
|
| 683 |
+
continue
|
| 684 |
+
slots, score = match
|
| 685 |
+
goal = _goal_from_instruction(raw, slots)
|
| 686 |
+
candidates.append(
|
| 687 |
+
(
|
| 688 |
+
score,
|
| 689 |
+
_system_context_priority(candidate_symbols, goal),
|
| 690 |
+
suffix_start,
|
| 691 |
+
len(pattern),
|
| 692 |
+
str(raw.get("name", "")),
|
| 693 |
+
goal,
|
| 694 |
+
candidate_symbols,
|
| 695 |
+
)
|
| 696 |
+
)
|
| 697 |
+
if not candidates:
|
| 698 |
+
raise ValueError("no instruction pattern matched")
|
| 699 |
+
candidates.sort(key=lambda item: (-item[0], -item[1], -item[2], -item[3], item[4]))
|
| 700 |
+
score, _system_priority, suffix_start, _pattern_length, name, goal, selected_symbols = candidates[0]
|
| 701 |
+
return {
|
| 702 |
+
"record_name": name,
|
| 703 |
+
"goal": goal,
|
| 704 |
+
"score": score,
|
| 705 |
+
"selected_suffix_start": suffix_start,
|
| 706 |
+
"selected_symbols": selected_symbols,
|
| 707 |
+
}
|
| 708 |
+
|
| 709 |
+
|
| 710 |
+
def _parse_context(artifact: dict[str, Any], symbols: list[str]) -> dict[str, Any]:
|
| 711 |
+
candidates: list[tuple[float, str, dict[str, Any]]] = []
|
| 712 |
+
for raw in artifact.get("contexts", {}).get("records", []) or []:
|
| 713 |
+
match = _match_pattern(
|
| 714 |
+
[str(item) for item in raw.get("pattern", [])],
|
| 715 |
+
symbols,
|
| 716 |
+
dict(raw.get("slot_examples", {}) or {}),
|
| 717 |
+
)
|
| 718 |
+
if match is None:
|
| 719 |
+
continue
|
| 720 |
+
slots, score = match
|
| 721 |
+
context = {}
|
| 722 |
+
for key, value in dict(raw.get("context_template", {}) or {}).items():
|
| 723 |
+
if isinstance(value, str) and value.startswith("$"):
|
| 724 |
+
context[str(key)] = slots[value[1:]]
|
| 725 |
+
else:
|
| 726 |
+
context[str(key)] = value
|
| 727 |
+
candidates.append((score, str(raw.get("name", "")), context))
|
| 728 |
+
if not candidates:
|
| 729 |
+
raise ValueError("no context pattern matched")
|
| 730 |
+
candidates.sort(key=lambda item: (-item[0], item[1]))
|
| 731 |
+
score, name, context = candidates[0]
|
| 732 |
+
return {"record_name": name, "context": context, "score": score}
|
| 733 |
+
|
| 734 |
+
|
| 735 |
+
def _compatible(learned: dict[str, Any], observed: dict[str, Any]) -> bool:
|
| 736 |
+
return all(key not in observed or observed[key] == value for key, value in learned.items())
|
| 737 |
+
|
| 738 |
+
|
| 739 |
+
def _overlap_score(learned: dict[str, Any], observed: dict[str, Any]) -> tuple[int, int]:
|
| 740 |
+
overlap = sum(1 for key, value in learned.items() if key in observed and observed[key] == value)
|
| 741 |
+
mismatch = sum(1 for key, value in learned.items() if key in observed and observed[key] != value)
|
| 742 |
+
return overlap, -mismatch
|
| 743 |
+
|
| 744 |
+
|
| 745 |
+
def _resolve_path(path: str, memory: dict[str, Any]) -> Any:
|
| 746 |
+
value: Any = memory
|
| 747 |
+
for part in path.split("."):
|
| 748 |
+
if isinstance(value, dict):
|
| 749 |
+
value = value[part]
|
| 750 |
+
elif isinstance(value, list):
|
| 751 |
+
value = value[int(part)]
|
| 752 |
+
else:
|
| 753 |
+
raise ValueError(f"missing working memory value: {path}")
|
| 754 |
+
return value
|
| 755 |
+
|
| 756 |
+
|
| 757 |
+
def _load_host_tool(spec: str):
|
| 758 |
+
if "=" not in spec:
|
| 759 |
+
if spec == "math.add":
|
| 760 |
+
return spec, lambda a, b: int(a) + int(b)
|
| 761 |
+
if spec == "math.power":
|
| 762 |
+
return spec, lambda base, exponent: int(base) ** int(exponent)
|
| 763 |
+
raise ValueError("custom host tool must use ARTIFACT_TOOL=module:function")
|
| 764 |
+
tool_name, target = (part.strip() for part in spec.split("=", 1))
|
| 765 |
+
module_name, attr_path = (part.strip() for part in target.split(":", 1))
|
| 766 |
+
cwd = str(Path.cwd())
|
| 767 |
+
if cwd not in sys.path:
|
| 768 |
+
sys.path.insert(0, cwd)
|
| 769 |
+
value: Any = importlib.import_module(module_name)
|
| 770 |
+
for part in attr_path.split("."):
|
| 771 |
+
value = getattr(value, part)
|
| 772 |
+
if not callable(value):
|
| 773 |
+
raise ValueError(f"custom host tool target is not callable: {target}")
|
| 774 |
+
return tool_name, value
|
| 775 |
+
|
| 776 |
+
|
| 777 |
+
def _builtin_host_tools() -> dict[str, Any]:
|
| 778 |
+
return {
|
| 779 |
+
"math.add": lambda a, b: int(a) + int(b),
|
| 780 |
+
"math.power": lambda base, exponent: int(base) ** int(exponent),
|
| 781 |
+
}
|
| 782 |
+
|
| 783 |
+
|
| 784 |
+
def _host_tools(specs: list[str]) -> dict[str, Any]:
|
| 785 |
+
tools = _builtin_host_tools()
|
| 786 |
+
for spec in specs:
|
| 787 |
+
name, func = _load_host_tool(spec)
|
| 788 |
+
tools[name] = func
|
| 789 |
+
return tools
|
| 790 |
+
|
| 791 |
+
|
| 792 |
+
def _select_procedure(artifact: dict[str, Any], goal: dict[str, Any]) -> dict[str, Any]:
|
| 793 |
+
candidates = []
|
| 794 |
+
for raw in artifact.get("cognition", {}).get("procedures", []) or []:
|
| 795 |
+
learned = dict(raw.get("goal", {}) or {})
|
| 796 |
+
if not _compatible(learned, goal):
|
| 797 |
+
continue
|
| 798 |
+
candidates.append((_overlap_score(learned, goal), str(raw.get("name", "")), raw))
|
| 799 |
+
if not candidates:
|
| 800 |
+
raise ValueError("no procedures recorded")
|
| 801 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1]))
|
| 802 |
+
return dict(candidates[0][2])
|
| 803 |
+
|
| 804 |
+
|
| 805 |
+
def _select_route(artifact: dict[str, Any], context: dict[str, Any]) -> dict[str, Any]:
|
| 806 |
+
candidates = []
|
| 807 |
+
for raw in artifact.get("autonomous_routes", {}).get("records", []) or []:
|
| 808 |
+
learned = dict(raw.get("context", {}) or {})
|
| 809 |
+
if not _compatible(learned, context):
|
| 810 |
+
continue
|
| 811 |
+
candidates.append((_overlap_score(learned, context), int(raw.get("uses", 0)), str(raw.get("name", "")), raw))
|
| 812 |
+
if not candidates:
|
| 813 |
+
raise ValueError("context does not contain learned autonomous route keys")
|
| 814 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2]))
|
| 815 |
+
raw = dict(candidates[0][3])
|
| 816 |
+
return {
|
| 817 |
+
"record_name": str(raw.get("name", "")),
|
| 818 |
+
"route": str(raw.get("route", "")),
|
| 819 |
+
"score": float(candidates[0][0][0]),
|
| 820 |
+
}
|
| 821 |
+
|
| 822 |
+
|
| 823 |
+
def _select_autonomous_goal(artifact: dict[str, Any], context: dict[str, Any]) -> dict[str, Any]:
|
| 824 |
+
candidates = []
|
| 825 |
+
for raw in artifact.get("autonomous_goals", {}).get("records", []) or []:
|
| 826 |
+
learned = dict(raw.get("context", {}) or {})
|
| 827 |
+
if not _compatible(learned, context):
|
| 828 |
+
continue
|
| 829 |
+
candidates.append((_overlap_score(learned, context), int(raw.get("uses", 0)), str(raw.get("name", "")), raw))
|
| 830 |
+
if not candidates:
|
| 831 |
+
raise ValueError("context does not contain learned autonomous goal keys")
|
| 832 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2]))
|
| 833 |
+
raw = dict(candidates[0][3])
|
| 834 |
+
goal = dict(raw.get("goal", {}) or {})
|
| 835 |
+
transfers = []
|
| 836 |
+
for source, target in dict(raw.get("goal_from_context", {}) or {}).items():
|
| 837 |
+
if source in context:
|
| 838 |
+
goal[str(target)] = context[source]
|
| 839 |
+
transfers.append([str(source), str(target), str(context[source])])
|
| 840 |
+
return {
|
| 841 |
+
"record_name": str(raw.get("name", "")),
|
| 842 |
+
"goal": goal,
|
| 843 |
+
"score": float(candidates[0][0][0]),
|
| 844 |
+
"context_transfers": transfers,
|
| 845 |
+
}
|
| 846 |
+
|
| 847 |
+
|
| 848 |
+
def _select_initiative(artifact: dict[str, Any], context: dict[str, Any]) -> dict[str, Any]:
|
| 849 |
+
candidates = []
|
| 850 |
+
for raw in artifact.get("initiative", {}).get("records", []) or []:
|
| 851 |
+
learned = dict(raw.get("context", {}) or {})
|
| 852 |
+
if not _compatible(learned, context):
|
| 853 |
+
continue
|
| 854 |
+
candidates.append((_overlap_score(learned, context), int(raw.get("uses", 0)), str(raw.get("name", "")), raw))
|
| 855 |
+
if not candidates:
|
| 856 |
+
raise ValueError("context does not contain learned initiative keys")
|
| 857 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2]))
|
| 858 |
+
raw = dict(candidates[0][3])
|
| 859 |
+
meaning = dict(raw.get("speech_meaning", {}) or {})
|
| 860 |
+
transfers = []
|
| 861 |
+
for source, target in dict(raw.get("meaning_from_context", {}) or {}).items():
|
| 862 |
+
if source not in context:
|
| 863 |
+
raise ValueError(f"initiative context missing meaning key: {source}")
|
| 864 |
+
meaning[str(target)] = context[source]
|
| 865 |
+
transfers.append([str(source), str(target), str(context[source])])
|
| 866 |
+
return {
|
| 867 |
+
"record_name": str(raw.get("name", "")),
|
| 868 |
+
"meaning": meaning,
|
| 869 |
+
"score": float(candidates[0][0][0]),
|
| 870 |
+
"context_transfers": transfers,
|
| 871 |
+
}
|
| 872 |
+
|
| 873 |
+
|
| 874 |
+
def _query_facts(artifact: dict[str, Any], arguments: dict[str, Any]) -> dict[str, Any]:
|
| 875 |
+
candidates = []
|
| 876 |
+
for raw in artifact.get("facts", {}).get("records", []) or []:
|
| 877 |
+
attrs = dict(raw.get("attributes", {}) or {})
|
| 878 |
+
score = _overlap_score(attrs, arguments)
|
| 879 |
+
if score[0] > 0:
|
| 880 |
+
candidates.append((score, str(raw.get("name", "")), attrs))
|
| 881 |
+
if not candidates:
|
| 882 |
+
raise ValueError("no facts matched")
|
| 883 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1]))
|
| 884 |
+
return candidates[0][2]
|
| 885 |
+
|
| 886 |
+
|
| 887 |
+
def _solve(artifact: dict[str, Any], goal: dict[str, Any], tools: dict[str, Any]) -> dict[str, Any]:
|
| 888 |
+
procedure = _select_procedure(artifact, goal)
|
| 889 |
+
working = dict(goal)
|
| 890 |
+
steps_out = []
|
| 891 |
+
for raw_step in procedure.get("steps", []) or []:
|
| 892 |
+
template = dict(raw_step.get("arguments", {}) or {})
|
| 893 |
+
arguments = {}
|
| 894 |
+
for key, value in template.items():
|
| 895 |
+
if isinstance(value, str) and value.startswith("$"):
|
| 896 |
+
arguments[str(key)] = _resolve_path(value[1:], working)
|
| 897 |
+
else:
|
| 898 |
+
arguments[str(key)] = value
|
| 899 |
+
tool_name = str(raw_step.get("tool_name", ""))
|
| 900 |
+
if tool_name == "fact.lookup":
|
| 901 |
+
result = _query_facts(artifact, arguments)
|
| 902 |
+
else:
|
| 903 |
+
func = tools.get(tool_name)
|
| 904 |
+
if func is None:
|
| 905 |
+
raise ValueError(f"tool not available: {tool_name}")
|
| 906 |
+
result = func(**arguments)
|
| 907 |
+
working[str(raw_step.get("output", ""))] = result
|
| 908 |
+
steps_out.append({"tool_name": tool_name, "arguments": arguments, "value": result})
|
| 909 |
+
final_value = steps_out[-1]["value"] if steps_out else None
|
| 910 |
+
return {
|
| 911 |
+
"procedure_name": str(procedure.get("name", "")),
|
| 912 |
+
"success": True,
|
| 913 |
+
"steps": steps_out,
|
| 914 |
+
"working_memory": working,
|
| 915 |
+
"final_value": final_value,
|
| 916 |
+
}
|
| 917 |
+
|
| 918 |
+
|
| 919 |
+
def _response_decision(artifact: dict[str, Any], goal: dict[str, Any], final_value: Any) -> dict[str, Any]:
|
| 920 |
+
candidates = []
|
| 921 |
+
for raw in artifact.get("response_policy", {}).get("records", []) or []:
|
| 922 |
+
learned = dict(raw.get("goal", {}) or {})
|
| 923 |
+
if not _compatible(learned, goal):
|
| 924 |
+
continue
|
| 925 |
+
candidates.append((_overlap_score(learned, goal), str(raw.get("name", "")), raw))
|
| 926 |
+
if not candidates:
|
| 927 |
+
raise ValueError("goal does not match any learned response policy")
|
| 928 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1]))
|
| 929 |
+
raw = dict(candidates[0][2])
|
| 930 |
+
meaning = dict(raw.get("response_meaning", {}) or {})
|
| 931 |
+
response_slot = str(raw.get("response_slot", ""))
|
| 932 |
+
if isinstance(final_value, dict):
|
| 933 |
+
meaning.update(final_value)
|
| 934 |
+
meaning[response_slot] = final_value
|
| 935 |
+
else:
|
| 936 |
+
meaning[response_slot] = final_value
|
| 937 |
+
for source, target in dict(raw.get("meaning_from_goal", {}) or {}).items():
|
| 938 |
+
if source in goal:
|
| 939 |
+
meaning[str(target)] = goal[source]
|
| 940 |
+
return {"record_name": str(raw.get("name", "")), "meaning": meaning, "response_slot": response_slot}
|
| 941 |
+
|
| 942 |
+
|
| 943 |
+
def _generate_speech(artifact: dict[str, Any], meaning: dict[str, Any], max_steps: int, avoid_texts: set[str]) -> dict[str, Any]:
|
| 944 |
+
records = list(artifact.get("speech", {}).get("records", []) or [])
|
| 945 |
+
learned_texts = set(str(text) for text in artifact.get("speech", {}).get("learned_texts", []) or [])
|
| 946 |
+
previous = "<START>"
|
| 947 |
+
trajectory = ""
|
| 948 |
+
trajectory_index = -1
|
| 949 |
+
symbols: list[str] = []
|
| 950 |
+
transition_names: list[str] = []
|
| 951 |
+
for _ in range(max_steps):
|
| 952 |
+
candidates = []
|
| 953 |
+
for raw in records:
|
| 954 |
+
if str(raw.get("previous", "")) != previous:
|
| 955 |
+
continue
|
| 956 |
+
learned = dict(raw.get("meaning", {}) or {})
|
| 957 |
+
if not _compatible(learned, meaning):
|
| 958 |
+
continue
|
| 959 |
+
record_trajectory = _transition_trajectory_name(raw)
|
| 960 |
+
record_index = _transition_index(raw)
|
| 961 |
+
if trajectory and record_trajectory != trajectory:
|
| 962 |
+
continue
|
| 963 |
+
if record_index >= 0 and record_index != trajectory_index + 1:
|
| 964 |
+
continue
|
| 965 |
+
candidates.append(
|
| 966 |
+
(
|
| 967 |
+
_overlap_score(learned, meaning),
|
| 968 |
+
_speech_trajectory_uses(records, record_trajectory),
|
| 969 |
+
int(raw.get("uses", 0)),
|
| 970 |
+
str(raw.get("name", "")),
|
| 971 |
+
raw,
|
| 972 |
+
)
|
| 973 |
+
)
|
| 974 |
+
if not candidates:
|
| 975 |
+
raise ValueError(f"no speech transition learned after: {previous}")
|
| 976 |
+
candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2], item[3]))
|
| 977 |
+
raw = dict(candidates[0][4])
|
| 978 |
+
if not trajectory:
|
| 979 |
+
trajectory = _transition_trajectory_name(raw)
|
| 980 |
+
trajectory_index = _transition_index(raw)
|
| 981 |
+
transition_names.append(str(raw.get("name", "")))
|
| 982 |
+
symbol = str(raw.get("symbol", ""))
|
| 983 |
+
if symbol == "<END>":
|
| 984 |
+
text = _render_text(symbols)
|
| 985 |
+
emitted = [text] if text in learned_texts or text in avoid_texts else []
|
| 986 |
+
return {
|
| 987 |
+
"record_name": transition_names[0] if transition_names else "",
|
| 988 |
+
"symbols": symbols,
|
| 989 |
+
"transition_names": transition_names,
|
| 990 |
+
"score": float(len(transition_names)),
|
| 991 |
+
"text": text,
|
| 992 |
+
"algorithm": "standalone-hdc-transition-speech",
|
| 993 |
+
"composed": False,
|
| 994 |
+
"trajectory_names": _speech_trajectory_names(records, transition_names),
|
| 995 |
+
"punctuation_symbols": [symbol for symbol in symbols if _is_punctuation(symbol)],
|
| 996 |
+
"emoji_symbols": [],
|
| 997 |
+
"slot_transfers": _slot_transfers(records, transition_names, meaning),
|
| 998 |
+
"replay_evidence": [],
|
| 999 |
+
"emitted_replay_evidence": emitted,
|
| 1000 |
+
"blocked_replay_evidence": [],
|
| 1001 |
+
}
|
| 1002 |
+
emitted_symbol = str(meaning[symbol[1:]]) if symbol.startswith("$") else symbol
|
| 1003 |
+
symbols.append(emitted_symbol)
|
| 1004 |
+
previous = symbol
|
| 1005 |
+
raise ValueError("speech rollout did not reach an end transition")
|
| 1006 |
+
|
| 1007 |
+
|
| 1008 |
+
def _transition_trajectory_name(raw: dict[str, Any]) -> str:
|
| 1009 |
+
name = str(raw.get("name", ""))
|
| 1010 |
+
if ":" not in name:
|
| 1011 |
+
return name
|
| 1012 |
+
return name.rsplit(":", 1)[0]
|
| 1013 |
+
|
| 1014 |
+
|
| 1015 |
+
def _transition_index(raw: dict[str, Any]) -> int:
|
| 1016 |
+
name = str(raw.get("name", ""))
|
| 1017 |
+
if ":" not in name:
|
| 1018 |
+
return -1
|
| 1019 |
+
try:
|
| 1020 |
+
return int(name.rsplit(":", 1)[1])
|
| 1021 |
+
except ValueError:
|
| 1022 |
+
return -1
|
| 1023 |
+
|
| 1024 |
+
|
| 1025 |
+
def _speech_trajectory_names(records: list[dict[str, Any]], transition_names: list[str]) -> list[str]:
|
| 1026 |
+
by_name = {str(raw.get("name", "")): raw for raw in records}
|
| 1027 |
+
names = []
|
| 1028 |
+
seen = set()
|
| 1029 |
+
for name in transition_names:
|
| 1030 |
+
raw = by_name.get(name)
|
| 1031 |
+
if raw is None or str(raw.get("symbol", "")) == "<END>":
|
| 1032 |
+
continue
|
| 1033 |
+
trajectory = _transition_trajectory_name(raw)
|
| 1034 |
+
if trajectory and trajectory not in seen:
|
| 1035 |
+
seen.add(trajectory)
|
| 1036 |
+
names.append(trajectory)
|
| 1037 |
+
return names
|
| 1038 |
+
|
| 1039 |
+
|
| 1040 |
+
def _speech_trajectory_uses(records: list[dict[str, Any]], trajectory: str) -> int:
|
| 1041 |
+
total = 0
|
| 1042 |
+
for raw in records:
|
| 1043 |
+
if _transition_trajectory_name(raw) == trajectory:
|
| 1044 |
+
total += int(raw.get("uses", 0))
|
| 1045 |
+
return total
|
| 1046 |
+
|
| 1047 |
+
|
| 1048 |
+
def _slot_transfers(records: list[dict[str, Any]], transition_names: list[str], meaning: dict[str, Any]) -> list[list[str]]:
|
| 1049 |
+
by_name = {str(raw.get("name", "")): raw for raw in records}
|
| 1050 |
+
transfers = []
|
| 1051 |
+
seen = set()
|
| 1052 |
+
for name in transition_names:
|
| 1053 |
+
symbol = str(by_name.get(name, {}).get("symbol", ""))
|
| 1054 |
+
if not symbol.startswith("$"):
|
| 1055 |
+
continue
|
| 1056 |
+
key = symbol[1:]
|
| 1057 |
+
if key in meaning and key not in seen:
|
| 1058 |
+
transfers.append([key, str(meaning[key])])
|
| 1059 |
+
seen.add(key)
|
| 1060 |
+
return transfers
|
| 1061 |
+
|
| 1062 |
+
|
| 1063 |
+
def _speech_transition_use_counts(records: list[dict[str, Any]], transition_names: list[str]) -> list[list[Any]]:
|
| 1064 |
+
by_name = {str(raw.get("name", "")): raw for raw in records}
|
| 1065 |
+
counts = []
|
| 1066 |
+
for name in transition_names:
|
| 1067 |
+
raw = by_name.get(name)
|
| 1068 |
+
if raw is not None:
|
| 1069 |
+
counts.append([name, int(raw.get("uses", 0))])
|
| 1070 |
+
return counts
|
| 1071 |
+
|
| 1072 |
+
|
| 1073 |
+
def _record_speech_usage(speech_payload: dict[str, Any], transition_names: list[str]) -> None:
|
| 1074 |
+
selected = set(str(name) for name in transition_names)
|
| 1075 |
+
for raw in speech_payload.get("records", []) or []:
|
| 1076 |
+
if str(raw.get("name", "")) in selected:
|
| 1077 |
+
raw["uses"] = int(raw.get("uses", 0)) + 1
|
| 1078 |
+
|
| 1079 |
+
|
| 1080 |
+
def _copy_exact(handle, out, length: int) -> bytes:
|
| 1081 |
+
payload = _read_exact(handle, length)
|
| 1082 |
+
out.write(payload)
|
| 1083 |
+
return payload
|
| 1084 |
+
|
| 1085 |
+
|
| 1086 |
+
def _copy_block(handle, out) -> bytes:
|
| 1087 |
+
length_payload = _read_exact(handle, 4)
|
| 1088 |
+
length = struct.unpack("<I", length_payload)[0]
|
| 1089 |
+
payload = _read_exact(handle, length)
|
| 1090 |
+
out.write(length_payload)
|
| 1091 |
+
out.write(payload)
|
| 1092 |
+
return payload
|
| 1093 |
+
|
| 1094 |
+
|
| 1095 |
+
def _write_block(out, payload: bytes) -> None:
|
| 1096 |
+
out.write(struct.pack("<I", len(payload)))
|
| 1097 |
+
out.write(payload)
|
| 1098 |
+
|
| 1099 |
+
|
| 1100 |
+
def _write_artifact_with_speech_payload(source_path: str | Path, target_path: str | Path, speech_payload: dict[str, Any]) -> None:
|
| 1101 |
+
target = Path(target_path)
|
| 1102 |
+
target.parent.mkdir(parents=True, exist_ok=True)
|
| 1103 |
+
with Path(source_path).open("rb") as handle, target.open("wb") as out:
|
| 1104 |
+
if _copy_exact(handle, out, len(MAGIC)) != MAGIC:
|
| 1105 |
+
raise ValueError("not an SLE CRA artifact")
|
| 1106 |
+
_copy_exact(handle, out, 4)
|
| 1107 |
+
_copy_block(handle, out)
|
| 1108 |
+
memory_header = _copy_exact(handle, out, 4 + 8 + 4)
|
| 1109 |
+
record_count = struct.unpack("<IQI", memory_header)[2]
|
| 1110 |
+
for _ in range(record_count):
|
| 1111 |
+
_copy_block(handle, out)
|
| 1112 |
+
_copy_block(handle, out)
|
| 1113 |
+
_copy_exact(handle, out, 4 + 4 + 8)
|
| 1114 |
+
for _ in range(5):
|
| 1115 |
+
_copy_block(handle, out)
|
| 1116 |
+
|
| 1117 |
+
marker = handle.read(len(TOOL_MAGIC))
|
| 1118 |
+
if not marker:
|
| 1119 |
+
return
|
| 1120 |
+
out.write(marker)
|
| 1121 |
+
if marker != TOOL_MAGIC:
|
| 1122 |
+
raise ValueError("unknown CRA extension section")
|
| 1123 |
+
tool_count_payload = _copy_exact(handle, out, 4)
|
| 1124 |
+
tool_count = struct.unpack("<I", tool_count_payload)[0]
|
| 1125 |
+
for _ in range(tool_count):
|
| 1126 |
+
_copy_block(handle, out)
|
| 1127 |
+
_copy_block(handle, out)
|
| 1128 |
+
_copy_block(handle, out)
|
| 1129 |
+
has_avoidance_payload = _copy_exact(handle, out, 1)
|
| 1130 |
+
if struct.unpack("<?", has_avoidance_payload)[0]:
|
| 1131 |
+
_copy_block(handle, out)
|
| 1132 |
+
|
| 1133 |
+
speech_written = False
|
| 1134 |
+
marker = handle.read(len(COGNITION_MAGIC))
|
| 1135 |
+
while marker:
|
| 1136 |
+
if len(marker) != len(COGNITION_MAGIC):
|
| 1137 |
+
raise ValueError("truncated CRA extension marker")
|
| 1138 |
+
out.write(marker)
|
| 1139 |
+
if marker == SPEECH_MAGIC:
|
| 1140 |
+
_read_block(handle)
|
| 1141 |
+
payload = json.dumps(
|
| 1142 |
+
speech_payload,
|
| 1143 |
+
sort_keys=True,
|
| 1144 |
+
separators=(",", ":"),
|
| 1145 |
+
).encode("utf-8")
|
| 1146 |
+
_write_block(out, payload)
|
| 1147 |
+
speech_written = True
|
| 1148 |
+
elif marker in (
|
| 1149 |
+
COGNITION_MAGIC,
|
| 1150 |
+
WORLD_MAGIC,
|
| 1151 |
+
LANGUAGE_MAGIC,
|
| 1152 |
+
DIALOGUE_MAGIC,
|
| 1153 |
+
INITIATIVE_MAGIC,
|
| 1154 |
+
AUTONOMOUS_GOAL_MAGIC,
|
| 1155 |
+
AUTONOMOUS_ROUTE_MAGIC,
|
| 1156 |
+
FACT_MAGIC,
|
| 1157 |
+
STREAM_MAGIC,
|
| 1158 |
+
ACOUSTIC_MAGIC,
|
| 1159 |
+
ACOUSTIC_SYMBOL_MAGIC,
|
| 1160 |
+
):
|
| 1161 |
+
_copy_block(handle, out)
|
| 1162 |
+
else:
|
| 1163 |
+
raise ValueError("unknown CRA extension section")
|
| 1164 |
+
marker = handle.read(len(COGNITION_MAGIC))
|
| 1165 |
+
if not speech_written:
|
| 1166 |
+
raise ValueError("artifact does not contain speech transition state")
|
| 1167 |
+
|
| 1168 |
+
|
| 1169 |
+
def _run_response(
|
| 1170 |
+
artifact: dict[str, Any],
|
| 1171 |
+
artifact_path: str,
|
| 1172 |
+
text: str,
|
| 1173 |
+
host_tool_specs: list[str],
|
| 1174 |
+
max_speech_steps: int,
|
| 1175 |
+
avoid_texts: set[str],
|
| 1176 |
+
canonicalize_stream: bool,
|
| 1177 |
+
) -> dict[str, Any]:
|
| 1178 |
+
tools = _host_tools(host_tool_specs)
|
| 1179 |
+
stream = _stream_evidence(text, artifact.get("stream", {}), canonicalize=canonicalize_stream)
|
| 1180 |
+
instruction = _parse_instruction_suffix(artifact, [str(chunk) for chunk in stream["chunks"]])
|
| 1181 |
+
cognition = _solve(artifact, dict(instruction["goal"]), tools)
|
| 1182 |
+
response = _response_decision(artifact, dict(instruction["goal"]), cognition["final_value"])
|
| 1183 |
+
speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts)
|
| 1184 |
+
steps = cognition["steps"]
|
| 1185 |
+
arguments = steps[0]["arguments"] if steps else {}
|
| 1186 |
+
tool_path = [step["tool_name"] for step in steps]
|
| 1187 |
+
emitted_replay = speech["emitted_replay_evidence"]
|
| 1188 |
+
return {
|
| 1189 |
+
"success": not bool(emitted_replay),
|
| 1190 |
+
"mode": "response",
|
| 1191 |
+
"artifact": artifact_path,
|
| 1192 |
+
"host_tools": host_tool_specs,
|
| 1193 |
+
"instruction_record": instruction["record_name"],
|
| 1194 |
+
"instruction_goal": instruction["goal"],
|
| 1195 |
+
"selected_suffix_start": instruction["selected_suffix_start"],
|
| 1196 |
+
"selected_symbols": instruction["selected_symbols"],
|
| 1197 |
+
"tool_path": tool_path,
|
| 1198 |
+
"arguments": arguments,
|
| 1199 |
+
"final_value": cognition["final_value"],
|
| 1200 |
+
"text": speech["text"],
|
| 1201 |
+
"algorithm": speech["algorithm"],
|
| 1202 |
+
"transition_names": speech["transition_names"],
|
| 1203 |
+
"punctuation_symbols": speech["punctuation_symbols"],
|
| 1204 |
+
"emoji_symbols": speech["emoji_symbols"],
|
| 1205 |
+
"slot_transfers": speech["slot_transfers"],
|
| 1206 |
+
"replay_evidence": speech["replay_evidence"],
|
| 1207 |
+
"emitted_replay_evidence": emitted_replay,
|
| 1208 |
+
"blocked_replay_evidence": speech["blocked_replay_evidence"],
|
| 1209 |
+
"stream": stream,
|
| 1210 |
+
}
|
| 1211 |
+
|
| 1212 |
+
|
| 1213 |
+
def _run_autonomous(
|
| 1214 |
+
artifact: dict[str, Any],
|
| 1215 |
+
artifact_path: str,
|
| 1216 |
+
text: str,
|
| 1217 |
+
host_tool_specs: list[str],
|
| 1218 |
+
max_speech_steps: int,
|
| 1219 |
+
avoid_texts: set[str],
|
| 1220 |
+
canonicalize_stream: bool,
|
| 1221 |
+
) -> dict[str, Any]:
|
| 1222 |
+
tools = _host_tools(host_tool_specs)
|
| 1223 |
+
stream = _stream_evidence(text, artifact.get("stream", {}), canonicalize=canonicalize_stream)
|
| 1224 |
+
context = _parse_context(artifact, [str(chunk) for chunk in stream["chunks"]])
|
| 1225 |
+
route = _select_route(artifact, dict(context["context"]))
|
| 1226 |
+
if route["route"] == "speech":
|
| 1227 |
+
initiative = _select_initiative(artifact, dict(context["context"]))
|
| 1228 |
+
speech = _generate_speech(artifact, dict(initiative["meaning"]), max_speech_steps, avoid_texts)
|
| 1229 |
+
emitted_replay = speech["emitted_replay_evidence"]
|
| 1230 |
+
return {
|
| 1231 |
+
"success": not bool(emitted_replay),
|
| 1232 |
+
"mode": "autonomous",
|
| 1233 |
+
"artifact": artifact_path,
|
| 1234 |
+
"host_tools": host_tool_specs,
|
| 1235 |
+
"route_record": route["record_name"],
|
| 1236 |
+
"route": route["route"],
|
| 1237 |
+
"context_record": context["record_name"],
|
| 1238 |
+
"context": context["context"],
|
| 1239 |
+
"context_score": context["score"],
|
| 1240 |
+
"decision": initiative["record_name"],
|
| 1241 |
+
"speech_meaning": initiative["meaning"],
|
| 1242 |
+
"initiative_context_transfers": initiative["context_transfers"],
|
| 1243 |
+
"goal_context_transfers": [],
|
| 1244 |
+
"tool_path": [],
|
| 1245 |
+
"arguments": [],
|
| 1246 |
+
"final_value": None,
|
| 1247 |
+
"text": speech["text"],
|
| 1248 |
+
"algorithm": speech["algorithm"],
|
| 1249 |
+
"transition_names": speech["transition_names"],
|
| 1250 |
+
"punctuation_symbols": speech["punctuation_symbols"],
|
| 1251 |
+
"emoji_symbols": speech["emoji_symbols"],
|
| 1252 |
+
"slot_transfers": speech["slot_transfers"],
|
| 1253 |
+
"replay_evidence": speech["replay_evidence"],
|
| 1254 |
+
"emitted_replay_evidence": emitted_replay,
|
| 1255 |
+
"blocked_replay_evidence": speech["blocked_replay_evidence"],
|
| 1256 |
+
"stream": stream,
|
| 1257 |
+
}
|
| 1258 |
+
if route["route"] != "action":
|
| 1259 |
+
raise ValueError(f"unsupported standalone autonomous route: {route['route']}")
|
| 1260 |
+
goal_decision = _select_autonomous_goal(artifact, dict(context["context"]))
|
| 1261 |
+
cognition = _solve(artifact, dict(goal_decision["goal"]), tools)
|
| 1262 |
+
response = _response_decision(artifact, dict(goal_decision["goal"]), cognition["final_value"])
|
| 1263 |
+
speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts)
|
| 1264 |
+
steps = cognition["steps"]
|
| 1265 |
+
emitted_replay = speech["emitted_replay_evidence"]
|
| 1266 |
+
return {
|
| 1267 |
+
"success": not bool(emitted_replay),
|
| 1268 |
+
"mode": "autonomous",
|
| 1269 |
+
"artifact": artifact_path,
|
| 1270 |
+
"host_tools": host_tool_specs,
|
| 1271 |
+
"route_record": route["record_name"],
|
| 1272 |
+
"route": route["route"],
|
| 1273 |
+
"context_record": context["record_name"],
|
| 1274 |
+
"context": context["context"],
|
| 1275 |
+
"context_score": context["score"],
|
| 1276 |
+
"decision": goal_decision["record_name"],
|
| 1277 |
+
"goal_context_transfers": goal_decision["context_transfers"],
|
| 1278 |
+
"tool_path": [step["tool_name"] for step in steps],
|
| 1279 |
+
"arguments": [step["arguments"] for step in steps],
|
| 1280 |
+
"final_value": cognition["final_value"],
|
| 1281 |
+
"text": speech["text"],
|
| 1282 |
+
"algorithm": speech["algorithm"],
|
| 1283 |
+
"transition_names": speech["transition_names"],
|
| 1284 |
+
"punctuation_symbols": speech["punctuation_symbols"],
|
| 1285 |
+
"emoji_symbols": speech["emoji_symbols"],
|
| 1286 |
+
"slot_transfers": speech["slot_transfers"],
|
| 1287 |
+
"replay_evidence": speech["replay_evidence"],
|
| 1288 |
+
"emitted_replay_evidence": emitted_replay,
|
| 1289 |
+
"blocked_replay_evidence": speech["blocked_replay_evidence"],
|
| 1290 |
+
"stream": stream,
|
| 1291 |
+
}
|
| 1292 |
+
|
| 1293 |
+
|
| 1294 |
+
def _run_autonomous_turns(
|
| 1295 |
+
artifact_path: str,
|
| 1296 |
+
text: str,
|
| 1297 |
+
host_tool_specs: list[str],
|
| 1298 |
+
max_speech_steps: int,
|
| 1299 |
+
avoid_texts: set[str],
|
| 1300 |
+
canonicalize_stream: bool,
|
| 1301 |
+
turns: int,
|
| 1302 |
+
save_artifact: str | None,
|
| 1303 |
+
diverse_speech_limit: int | None,
|
| 1304 |
+
punctuation_floor: int | None,
|
| 1305 |
+
) -> dict[str, Any]:
|
| 1306 |
+
if turns < 1:
|
| 1307 |
+
raise ValueError("--turns must be at least 1")
|
| 1308 |
+
if diverse_speech_limit is not None and diverse_speech_limit < 1:
|
| 1309 |
+
raise ValueError("--diverse-speech-limit must be positive")
|
| 1310 |
+
if punctuation_floor is not None and punctuation_floor < 0:
|
| 1311 |
+
raise ValueError("--punctuation-floor must be non-negative")
|
| 1312 |
+
turn_records = []
|
| 1313 |
+
with tempfile.TemporaryDirectory() as tmp:
|
| 1314 |
+
current_path = Path(artifact_path)
|
| 1315 |
+
for turn_index in range(turns):
|
| 1316 |
+
artifact = _load_artifact(current_path)
|
| 1317 |
+
payload = _run_autonomous(
|
| 1318 |
+
artifact,
|
| 1319 |
+
str(current_path),
|
| 1320 |
+
text,
|
| 1321 |
+
host_tool_specs,
|
| 1322 |
+
max_speech_steps,
|
| 1323 |
+
avoid_texts,
|
| 1324 |
+
canonicalize_stream,
|
| 1325 |
+
)
|
| 1326 |
+
records = list(artifact.get("speech", {}).get("records", []) or [])
|
| 1327 |
+
before = _speech_transition_use_counts(records, payload.get("transition_names", []))
|
| 1328 |
+
_record_speech_usage(artifact.get("speech", {}), payload.get("transition_names", []))
|
| 1329 |
+
after = _speech_transition_use_counts(
|
| 1330 |
+
list(artifact.get("speech", {}).get("records", []) or []),
|
| 1331 |
+
payload.get("transition_names", []),
|
| 1332 |
+
)
|
| 1333 |
+
payload["turn_index"] = turn_index
|
| 1334 |
+
payload["transition_use_counts_before"] = before
|
| 1335 |
+
payload["transition_use_counts_after"] = after
|
| 1336 |
+
turn_records.append(payload)
|
| 1337 |
+
|
| 1338 |
+
if turn_index == turns - 1 and save_artifact:
|
| 1339 |
+
next_path = Path(save_artifact)
|
| 1340 |
+
else:
|
| 1341 |
+
next_path = Path(tmp) / f"turn_{turn_index}.cra"
|
| 1342 |
+
_write_artifact_with_speech_payload(current_path, next_path, artifact.get("speech", {}))
|
| 1343 |
+
current_path = next_path
|
| 1344 |
+
|
| 1345 |
+
texts = [str(turn.get("text", "")) for turn in turn_records]
|
| 1346 |
+
emitted_replay = []
|
| 1347 |
+
blocked_replay = []
|
| 1348 |
+
punctuation = []
|
| 1349 |
+
for turn in turn_records:
|
| 1350 |
+
emitted_replay.extend(turn.get("emitted_replay_evidence", []) or [])
|
| 1351 |
+
blocked_replay.extend(turn.get("blocked_replay_evidence", []) or [])
|
| 1352 |
+
punctuation.extend(str(symbol) for symbol in turn.get("punctuation_symbols", []) or [])
|
| 1353 |
+
distinct_punctuation = sorted(set(punctuation))
|
| 1354 |
+
punctuation_success = punctuation_floor is None or len(distinct_punctuation) >= punctuation_floor
|
| 1355 |
+
return {
|
| 1356 |
+
"success": all(bool(turn.get("success")) for turn in turn_records)
|
| 1357 |
+
and not bool(emitted_replay)
|
| 1358 |
+
and punctuation_success,
|
| 1359 |
+
"mode": "autonomous",
|
| 1360 |
+
"artifact": artifact_path,
|
| 1361 |
+
"host_tools": host_tool_specs,
|
| 1362 |
+
"turn_count": len(turn_records),
|
| 1363 |
+
"write_reload": turns > 1,
|
| 1364 |
+
"saved_artifact": str(save_artifact or ""),
|
| 1365 |
+
"diverse_speech_limit": diverse_speech_limit,
|
| 1366 |
+
"punctuation_floor": punctuation_floor,
|
| 1367 |
+
"texts": texts,
|
| 1368 |
+
"text": texts[-1] if texts else "",
|
| 1369 |
+
"unique_text_count": len(set(texts)),
|
| 1370 |
+
"distinct_punctuation_symbols": distinct_punctuation,
|
| 1371 |
+
"punctuation_success": punctuation_success,
|
| 1372 |
+
"turn_records": turn_records,
|
| 1373 |
+
"emitted_replay_evidence": emitted_replay,
|
| 1374 |
+
"blocked_replay_evidence": blocked_replay,
|
| 1375 |
+
}
|
| 1376 |
+
|
| 1377 |
+
|
| 1378 |
+
def _instruction_segments(symbols: list[str]) -> list[list[str]]:
|
| 1379 |
+
segments: list[list[str]] = []
|
| 1380 |
+
current: list[str] = []
|
| 1381 |
+
for symbol in symbols:
|
| 1382 |
+
current.append(symbol)
|
| 1383 |
+
if (
|
| 1384 |
+
symbol == ":"
|
| 1385 |
+
and len(current) == 2
|
| 1386 |
+
and str(current[0]).lower() in ROLE_LABELS
|
| 1387 |
+
):
|
| 1388 |
+
segments.append(current)
|
| 1389 |
+
current = []
|
| 1390 |
+
elif symbol in TERMINAL_PUNCTUATION:
|
| 1391 |
+
segments.append(current)
|
| 1392 |
+
current = []
|
| 1393 |
+
if current:
|
| 1394 |
+
segments.append(current)
|
| 1395 |
+
return [segment for segment in segments if segment]
|
| 1396 |
+
|
| 1397 |
+
|
| 1398 |
+
def _history_instruction_segments(symbols: list[str]) -> list[list[str]]:
|
| 1399 |
+
raw_segments = _instruction_segments(symbols)
|
| 1400 |
+
segments: list[list[str]] = []
|
| 1401 |
+
index = 0
|
| 1402 |
+
while index < len(raw_segments):
|
| 1403 |
+
segment = raw_segments[index]
|
| 1404 |
+
if segment and segment[-1] == ":" and index + 1 < len(raw_segments):
|
| 1405 |
+
segments.append(segment + raw_segments[index + 1])
|
| 1406 |
+
index += 2
|
| 1407 |
+
else:
|
| 1408 |
+
segments.append(segment)
|
| 1409 |
+
index += 1
|
| 1410 |
+
return segments
|
| 1411 |
+
|
| 1412 |
+
|
| 1413 |
+
def _segment_role(segment: list[str]) -> str:
|
| 1414 |
+
if len(segment) >= 2 and segment[1] == ":":
|
| 1415 |
+
role = str(segment[0]).lower()
|
| 1416 |
+
if role in ROLE_LABELS:
|
| 1417 |
+
return role
|
| 1418 |
+
return ""
|
| 1419 |
+
|
| 1420 |
+
|
| 1421 |
+
def _history_search_order(segments: list[list[str]]) -> list[int]:
|
| 1422 |
+
user_indices = [index for index, segment in enumerate(segments) if _segment_role(segment) == "user"]
|
| 1423 |
+
if user_indices:
|
| 1424 |
+
return list(reversed(user_indices))
|
| 1425 |
+
non_assistant_indices = [
|
| 1426 |
+
index
|
| 1427 |
+
for index, segment in enumerate(segments)
|
| 1428 |
+
if _segment_role(segment) != "assistant"
|
| 1429 |
+
]
|
| 1430 |
+
if non_assistant_indices:
|
| 1431 |
+
return list(reversed(non_assistant_indices))
|
| 1432 |
+
return list(range(len(segments) - 1, -1, -1))
|
| 1433 |
+
|
| 1434 |
+
|
| 1435 |
+
def _run_history(
|
| 1436 |
+
artifact: dict[str, Any],
|
| 1437 |
+
artifact_path: str,
|
| 1438 |
+
text: str,
|
| 1439 |
+
host_tool_specs: list[str],
|
| 1440 |
+
max_speech_steps: int,
|
| 1441 |
+
avoid_texts: set[str],
|
| 1442 |
+
canonicalize_stream: bool,
|
| 1443 |
+
) -> dict[str, Any]:
|
| 1444 |
+
tools = _host_tools(host_tool_specs)
|
| 1445 |
+
stream = _stream_evidence(text, artifact.get("stream", {}), canonicalize=canonicalize_stream)
|
| 1446 |
+
segments = _history_instruction_segments([str(chunk) for chunk in stream["chunks"]])
|
| 1447 |
+
parse_errors: list[str] = []
|
| 1448 |
+
for segment_index in _history_search_order(segments):
|
| 1449 |
+
segment = segments[segment_index]
|
| 1450 |
+
for suffix_start in range(len(segment)):
|
| 1451 |
+
candidate = segment[suffix_start:]
|
| 1452 |
+
if not candidate or _is_punctuation(candidate[0]):
|
| 1453 |
+
continue
|
| 1454 |
+
try:
|
| 1455 |
+
instruction = _parse_instruction(artifact, candidate)
|
| 1456 |
+
cognition = _solve(artifact, dict(instruction["goal"]), tools)
|
| 1457 |
+
response = _response_decision(artifact, dict(instruction["goal"]), cognition["final_value"])
|
| 1458 |
+
speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts)
|
| 1459 |
+
except ValueError as exc:
|
| 1460 |
+
parse_errors.append(str(exc))
|
| 1461 |
+
continue
|
| 1462 |
+
steps = cognition["steps"]
|
| 1463 |
+
emitted_replay = speech["emitted_replay_evidence"]
|
| 1464 |
+
return {
|
| 1465 |
+
"success": not bool(emitted_replay),
|
| 1466 |
+
"mode": "history",
|
| 1467 |
+
"artifact": artifact_path,
|
| 1468 |
+
"host_tools": host_tool_specs,
|
| 1469 |
+
"history_segments": segments,
|
| 1470 |
+
"selected_segment_index": segment_index,
|
| 1471 |
+
"selected_suffix_start": suffix_start,
|
| 1472 |
+
"selected_symbols": candidate,
|
| 1473 |
+
"instruction_record": instruction["record_name"],
|
| 1474 |
+
"instruction_goal": instruction["goal"],
|
| 1475 |
+
"tool_path": [step["tool_name"] for step in steps],
|
| 1476 |
+
"arguments": [step["arguments"] for step in steps],
|
| 1477 |
+
"final_value": cognition["final_value"],
|
| 1478 |
+
"text": speech["text"],
|
| 1479 |
+
"algorithm": speech["algorithm"],
|
| 1480 |
+
"transition_names": speech["transition_names"],
|
| 1481 |
+
"punctuation_symbols": speech["punctuation_symbols"],
|
| 1482 |
+
"emoji_symbols": speech["emoji_symbols"],
|
| 1483 |
+
"slot_transfers": speech["slot_transfers"],
|
| 1484 |
+
"replay_evidence": speech["replay_evidence"],
|
| 1485 |
+
"emitted_replay_evidence": emitted_replay,
|
| 1486 |
+
"blocked_replay_evidence": speech["blocked_replay_evidence"],
|
| 1487 |
+
"stream": stream,
|
| 1488 |
+
}
|
| 1489 |
+
error = "no parseable instruction found in conversation history"
|
| 1490 |
+
if parse_errors:
|
| 1491 |
+
error = f"{error}: {parse_errors[-1]}"
|
| 1492 |
+
return {
|
| 1493 |
+
"success": False,
|
| 1494 |
+
"mode": "history",
|
| 1495 |
+
"artifact": artifact_path,
|
| 1496 |
+
"host_tools": host_tool_specs,
|
| 1497 |
+
"history_segments": segments,
|
| 1498 |
+
"selected_segment_index": -1,
|
| 1499 |
+
"selected_suffix_start": -1,
|
| 1500 |
+
"selected_symbols": [],
|
| 1501 |
+
"stream": stream,
|
| 1502 |
+
"error": error,
|
| 1503 |
+
}
|
| 1504 |
+
|
| 1505 |
+
|
| 1506 |
+
def _run_acoustic(
|
| 1507 |
+
artifact: dict[str, Any],
|
| 1508 |
+
artifact_path: str,
|
| 1509 |
+
audio_samples: list[float],
|
| 1510 |
+
host_tool_specs: list[str],
|
| 1511 |
+
max_speech_steps: int,
|
| 1512 |
+
avoid_texts: set[str],
|
| 1513 |
+
min_symbol_score: float,
|
| 1514 |
+
) -> dict[str, Any]:
|
| 1515 |
+
tools = _host_tools(host_tool_specs)
|
| 1516 |
+
stream = _acoustic_stream_evidence(audio_samples, artifact.get("acoustic", {}) or {})
|
| 1517 |
+
symbol_run = _decode_acoustic_symbols(
|
| 1518 |
+
artifact,
|
| 1519 |
+
audio_samples,
|
| 1520 |
+
list(stream["chunks"]),
|
| 1521 |
+
float(min_symbol_score),
|
| 1522 |
+
)
|
| 1523 |
+
instruction = _parse_instruction(artifact, [str(symbol) for symbol in symbol_run["symbols"]])
|
| 1524 |
+
cognition = _solve(artifact, dict(instruction["goal"]), tools)
|
| 1525 |
+
response = _response_decision(artifact, dict(instruction["goal"]), cognition["final_value"])
|
| 1526 |
+
speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts)
|
| 1527 |
+
steps = cognition["steps"]
|
| 1528 |
+
emitted_replay = speech["emitted_replay_evidence"]
|
| 1529 |
+
return {
|
| 1530 |
+
"success": not bool(emitted_replay),
|
| 1531 |
+
"mode": "acoustic",
|
| 1532 |
+
"artifact": artifact_path,
|
| 1533 |
+
"host_tools": host_tool_specs,
|
| 1534 |
+
"symbols": symbol_run["symbols"],
|
| 1535 |
+
"symbol_algorithm": symbol_run["algorithm"],
|
| 1536 |
+
"symbol_matches": symbol_run["matches"],
|
| 1537 |
+
"instruction_record": instruction["record_name"],
|
| 1538 |
+
"instruction_goal": instruction["goal"],
|
| 1539 |
+
"tool_path": [step["tool_name"] for step in steps],
|
| 1540 |
+
"arguments": [step["arguments"] for step in steps],
|
| 1541 |
+
"final_value": cognition["final_value"],
|
| 1542 |
+
"text": speech["text"],
|
| 1543 |
+
"texts": [speech["text"]],
|
| 1544 |
+
"algorithm": speech["algorithm"],
|
| 1545 |
+
"transition_names": speech["transition_names"],
|
| 1546 |
+
"punctuation_symbols": speech["punctuation_symbols"],
|
| 1547 |
+
"emoji_symbols": speech["emoji_symbols"],
|
| 1548 |
+
"slot_transfers": speech["slot_transfers"],
|
| 1549 |
+
"replay_evidence": speech["replay_evidence"],
|
| 1550 |
+
"emitted_replay_evidence": emitted_replay,
|
| 1551 |
+
"blocked_replay_evidence": speech["blocked_replay_evidence"],
|
| 1552 |
+
"stream": stream,
|
| 1553 |
+
}
|
| 1554 |
+
|
| 1555 |
+
|
| 1556 |
+
def _openai_message_content(content: Any) -> str:
|
| 1557 |
+
if isinstance(content, str):
|
| 1558 |
+
return content
|
| 1559 |
+
if isinstance(content, list):
|
| 1560 |
+
parts: list[str] = []
|
| 1561 |
+
for item in content:
|
| 1562 |
+
if isinstance(item, str):
|
| 1563 |
+
parts.append(item)
|
| 1564 |
+
elif isinstance(item, dict):
|
| 1565 |
+
if "text" in item:
|
| 1566 |
+
parts.append(str(item["text"]))
|
| 1567 |
+
elif "content" in item:
|
| 1568 |
+
parts.append(str(item["content"]))
|
| 1569 |
+
return " ".join(part for part in parts if part)
|
| 1570 |
+
if content is None:
|
| 1571 |
+
return ""
|
| 1572 |
+
return str(content)
|
| 1573 |
+
|
| 1574 |
+
|
| 1575 |
+
def _openai_role_label(role: Any) -> str:
|
| 1576 |
+
name = str(role or "user").strip().lower()
|
| 1577 |
+
if name not in ROLE_LABELS:
|
| 1578 |
+
name = "user"
|
| 1579 |
+
return name.capitalize()
|
| 1580 |
+
|
| 1581 |
+
|
| 1582 |
+
def _openai_messages_to_history_text(messages: Any) -> str:
|
| 1583 |
+
if not isinstance(messages, list):
|
| 1584 |
+
raise ValueError("OpenAI chat completion payload requires messages as a list")
|
| 1585 |
+
segments = []
|
| 1586 |
+
for raw in messages:
|
| 1587 |
+
if not isinstance(raw, dict):
|
| 1588 |
+
raise ValueError("OpenAI chat message must be an object")
|
| 1589 |
+
content = _openai_message_content(raw.get("content", ""))
|
| 1590 |
+
if not content:
|
| 1591 |
+
continue
|
| 1592 |
+
segments.append(f"{_openai_role_label(raw.get('role', 'user'))}: {content}")
|
| 1593 |
+
if not segments:
|
| 1594 |
+
raise ValueError("OpenAI chat completion payload requires at least one non-empty message")
|
| 1595 |
+
return " ".join(segments)
|
| 1596 |
+
|
| 1597 |
+
|
| 1598 |
+
def _extend_tool_specs(specs: list[str], value: Any) -> None:
|
| 1599 |
+
if value is None:
|
| 1600 |
+
return
|
| 1601 |
+
if isinstance(value, str):
|
| 1602 |
+
if value:
|
| 1603 |
+
specs.append(value)
|
| 1604 |
+
return
|
| 1605 |
+
if isinstance(value, list):
|
| 1606 |
+
for item in value:
|
| 1607 |
+
if item:
|
| 1608 |
+
specs.append(str(item))
|
| 1609 |
+
|
| 1610 |
+
|
| 1611 |
+
def _openai_host_tool_specs(request: dict[str, Any]) -> list[str]:
|
| 1612 |
+
specs: list[str] = []
|
| 1613 |
+
_extend_tool_specs(specs, request.get("host_tools"))
|
| 1614 |
+
_extend_tool_specs(specs, request.get("saccadic_host_tools"))
|
| 1615 |
+
saccadic_options = request.get("saccadic")
|
| 1616 |
+
if isinstance(saccadic_options, dict):
|
| 1617 |
+
_extend_tool_specs(specs, saccadic_options.get("host_tools"))
|
| 1618 |
+
tools = request.get("tools")
|
| 1619 |
+
if isinstance(tools, list):
|
| 1620 |
+
for tool in tools:
|
| 1621 |
+
if not isinstance(tool, dict):
|
| 1622 |
+
continue
|
| 1623 |
+
spec = tool.get("saccadic_host_tool") or tool.get("x_saccadic_host_tool")
|
| 1624 |
+
if spec:
|
| 1625 |
+
specs.append(str(spec))
|
| 1626 |
+
return specs
|
| 1627 |
+
|
| 1628 |
+
|
| 1629 |
+
def _openai_request_option(request: dict[str, Any], key: str, default: Any) -> Any:
|
| 1630 |
+
if key in request:
|
| 1631 |
+
return request[key]
|
| 1632 |
+
saccadic_options = request.get("saccadic")
|
| 1633 |
+
if isinstance(saccadic_options, dict) and key in saccadic_options:
|
| 1634 |
+
return saccadic_options[key]
|
| 1635 |
+
return default
|
| 1636 |
+
|
| 1637 |
+
|
| 1638 |
+
def _run_openai_chat_completion(
|
| 1639 |
+
artifact: dict[str, Any],
|
| 1640 |
+
artifact_path: str,
|
| 1641 |
+
request: dict[str, Any],
|
| 1642 |
+
default_max_speech_steps: int,
|
| 1643 |
+
) -> dict[str, Any]:
|
| 1644 |
+
text = _openai_messages_to_history_text(request.get("messages"))
|
| 1645 |
+
payload = _run_history(
|
| 1646 |
+
artifact,
|
| 1647 |
+
artifact_path,
|
| 1648 |
+
text,
|
| 1649 |
+
_openai_host_tool_specs(request),
|
| 1650 |
+
int(_openai_request_option(request, "max_speech_steps", default_max_speech_steps)),
|
| 1651 |
+
set(str(item) for item in _openai_request_option(request, "avoid_texts", []) or []),
|
| 1652 |
+
bool(_openai_request_option(request, "canonicalize_stream", False)),
|
| 1653 |
+
)
|
| 1654 |
+
payload["service"] = "sle-artifact-http"
|
| 1655 |
+
return payload
|
| 1656 |
+
|
| 1657 |
+
|
| 1658 |
+
def _openai_completion_id(payload: dict[str, Any]) -> str:
|
| 1659 |
+
text = str(payload.get("text", ""))
|
| 1660 |
+
value = _stable_text_hash(text, int(payload.get("memory_seed", 0) or 0))
|
| 1661 |
+
return f"chatcmpl-saccadic-{value:016x}"
|
| 1662 |
+
|
| 1663 |
+
|
| 1664 |
+
def _openai_completion_response(model: str, payload: dict[str, Any]) -> dict[str, Any]:
|
| 1665 |
+
content = str(payload.get("text", ""))
|
| 1666 |
+
return {
|
| 1667 |
+
"id": _openai_completion_id(payload),
|
| 1668 |
+
"object": "chat.completion",
|
| 1669 |
+
"created": int(time.time()),
|
| 1670 |
+
"model": model,
|
| 1671 |
+
"choices": [
|
| 1672 |
+
{
|
| 1673 |
+
"index": 0,
|
| 1674 |
+
"message": {"role": "assistant", "content": content},
|
| 1675 |
+
"finish_reason": "stop" if payload.get("success") else "error",
|
| 1676 |
+
}
|
| 1677 |
+
],
|
| 1678 |
+
"usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},
|
| 1679 |
+
"saccadic": payload,
|
| 1680 |
+
}
|
| 1681 |
+
|
| 1682 |
+
|
| 1683 |
+
def _openai_content_chunks(text: str) -> list[str]:
|
| 1684 |
+
chunks: list[str] = []
|
| 1685 |
+
current = ""
|
| 1686 |
+
for character in text:
|
| 1687 |
+
current += character
|
| 1688 |
+
if character.isspace():
|
| 1689 |
+
chunks.append(current)
|
| 1690 |
+
current = ""
|
| 1691 |
+
if current:
|
| 1692 |
+
chunks.append(current)
|
| 1693 |
+
return chunks
|
| 1694 |
+
|
| 1695 |
+
|
| 1696 |
+
def _openai_stream_response(model: str, payload: dict[str, Any]) -> str:
|
| 1697 |
+
completion_id = _openai_completion_id(payload)
|
| 1698 |
+
created = int(time.time())
|
| 1699 |
+
events: list[dict[str, Any]] = [
|
| 1700 |
+
{
|
| 1701 |
+
"id": completion_id,
|
| 1702 |
+
"object": "chat.completion.chunk",
|
| 1703 |
+
"created": created,
|
| 1704 |
+
"model": model,
|
| 1705 |
+
"choices": [{"index": 0, "delta": {"role": "assistant"}, "finish_reason": None}],
|
| 1706 |
+
}
|
| 1707 |
+
]
|
| 1708 |
+
for chunk in _openai_content_chunks(str(payload.get("text", ""))):
|
| 1709 |
+
events.append(
|
| 1710 |
+
{
|
| 1711 |
+
"id": completion_id,
|
| 1712 |
+
"object": "chat.completion.chunk",
|
| 1713 |
+
"created": created,
|
| 1714 |
+
"model": model,
|
| 1715 |
+
"choices": [{"index": 0, "delta": {"content": chunk}, "finish_reason": None}],
|
| 1716 |
+
}
|
| 1717 |
+
)
|
| 1718 |
+
events.append(
|
| 1719 |
+
{
|
| 1720 |
+
"id": completion_id,
|
| 1721 |
+
"object": "chat.completion.chunk",
|
| 1722 |
+
"created": created,
|
| 1723 |
+
"model": model,
|
| 1724 |
+
"choices": [{"index": 0, "delta": {}, "finish_reason": None}],
|
| 1725 |
+
"saccadic": payload,
|
| 1726 |
+
}
|
| 1727 |
+
)
|
| 1728 |
+
events.append(
|
| 1729 |
+
{
|
| 1730 |
+
"id": completion_id,
|
| 1731 |
+
"object": "chat.completion.chunk",
|
| 1732 |
+
"created": created,
|
| 1733 |
+
"model": model,
|
| 1734 |
+
"choices": [
|
| 1735 |
+
{
|
| 1736 |
+
"index": 0,
|
| 1737 |
+
"delta": {},
|
| 1738 |
+
"finish_reason": "stop" if payload.get("success") else "error",
|
| 1739 |
+
}
|
| 1740 |
+
],
|
| 1741 |
+
}
|
| 1742 |
+
)
|
| 1743 |
+
body = "".join(
|
| 1744 |
+
f"data: {json.dumps(event, ensure_ascii=True, sort_keys=True, separators=(',', ':'))}\n\n"
|
| 1745 |
+
for event in events
|
| 1746 |
+
)
|
| 1747 |
+
return body + "data: [DONE]\n\n"
|
| 1748 |
+
|
| 1749 |
+
|
| 1750 |
+
def _run_artifact(args: argparse.Namespace) -> int:
|
| 1751 |
+
if getattr(args, "compose_speech", False) or getattr(args, "prefer_composed", False):
|
| 1752 |
+
raise ValueError("standalone Python Spark does not support composed speech flags yet")
|
| 1753 |
+
if getattr(args, "recover_on_failure", False):
|
| 1754 |
+
raise ValueError("standalone Python Spark does not support autonomous recovery flags yet")
|
| 1755 |
+
artifact = _load_artifact(args.artifact)
|
| 1756 |
+
avoid_texts = set(str(item) for item in args.avoid_text)
|
| 1757 |
+
if args.mode == "response":
|
| 1758 |
+
payload = _run_response(
|
| 1759 |
+
artifact,
|
| 1760 |
+
str(args.artifact),
|
| 1761 |
+
str(args.text),
|
| 1762 |
+
list(args.host_tool),
|
| 1763 |
+
int(args.max_speech_steps),
|
| 1764 |
+
avoid_texts,
|
| 1765 |
+
bool(args.canonicalize_stream),
|
| 1766 |
+
)
|
| 1767 |
+
elif args.mode == "history":
|
| 1768 |
+
payload = _run_history(
|
| 1769 |
+
artifact,
|
| 1770 |
+
str(args.artifact),
|
| 1771 |
+
str(args.text),
|
| 1772 |
+
list(args.host_tool),
|
| 1773 |
+
int(args.max_speech_steps),
|
| 1774 |
+
avoid_texts,
|
| 1775 |
+
bool(args.canonicalize_stream),
|
| 1776 |
+
)
|
| 1777 |
+
elif args.mode == "autonomous":
|
| 1778 |
+
if int(args.turns) > 1:
|
| 1779 |
+
payload = _run_autonomous_turns(
|
| 1780 |
+
str(args.artifact),
|
| 1781 |
+
str(args.text),
|
| 1782 |
+
list(args.host_tool),
|
| 1783 |
+
int(args.max_speech_steps),
|
| 1784 |
+
avoid_texts,
|
| 1785 |
+
bool(args.canonicalize_stream),
|
| 1786 |
+
int(args.turns),
|
| 1787 |
+
args.save_artifact,
|
| 1788 |
+
args.diverse_speech_limit,
|
| 1789 |
+
args.punctuation_floor,
|
| 1790 |
+
)
|
| 1791 |
+
else:
|
| 1792 |
+
payload = _run_autonomous(
|
| 1793 |
+
artifact,
|
| 1794 |
+
str(args.artifact),
|
| 1795 |
+
str(args.text),
|
| 1796 |
+
list(args.host_tool),
|
| 1797 |
+
int(args.max_speech_steps),
|
| 1798 |
+
avoid_texts,
|
| 1799 |
+
bool(args.canonicalize_stream),
|
| 1800 |
+
)
|
| 1801 |
+
elif args.mode == "acoustic":
|
| 1802 |
+
payload = _run_acoustic(
|
| 1803 |
+
artifact,
|
| 1804 |
+
str(args.artifact),
|
| 1805 |
+
_parse_audio_samples(args.audio_samples),
|
| 1806 |
+
list(args.host_tool),
|
| 1807 |
+
int(args.max_speech_steps),
|
| 1808 |
+
avoid_texts,
|
| 1809 |
+
float(args.audio_symbol_score_floor),
|
| 1810 |
+
)
|
| 1811 |
+
else:
|
| 1812 |
+
raise ValueError("standalone Python Spark currently supports response, history, autonomous, and acoustic modes")
|
| 1813 |
+
print(json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":")))
|
| 1814 |
+
return 0 if payload.get("success") else 1
|
| 1815 |
+
|
| 1816 |
+
|
| 1817 |
+
def _serve_artifact(args: argparse.Namespace) -> int:
|
| 1818 |
+
artifact = _load_artifact(args.artifact)
|
| 1819 |
+
if args.smoke_once:
|
| 1820 |
+
stream = _stream_evidence("Saccadic service runtime smoke.", artifact.get("stream", {}))
|
| 1821 |
+
payload = {
|
| 1822 |
+
"success": True,
|
| 1823 |
+
"service": "sle-artifact-http",
|
| 1824 |
+
"artifact": str(args.artifact),
|
| 1825 |
+
"endpoint": "POST /run",
|
| 1826 |
+
"stream_dynamics": stream,
|
| 1827 |
+
}
|
| 1828 |
+
print(json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":")))
|
| 1829 |
+
return 0
|
| 1830 |
+
|
| 1831 |
+
artifact_path = str(args.artifact)
|
| 1832 |
+
|
| 1833 |
+
class Handler(BaseHTTPRequestHandler):
|
| 1834 |
+
def log_message(self, _format: str, *values: Any) -> None:
|
| 1835 |
+
return
|
| 1836 |
+
|
| 1837 |
+
def do_POST(self) -> None:
|
| 1838 |
+
length = int(self.headers.get("Content-Length", "0") or 0)
|
| 1839 |
+
request = json.loads(self.rfile.read(length).decode("utf-8")) if length else {}
|
| 1840 |
+
try:
|
| 1841 |
+
content_type = "application/json"
|
| 1842 |
+
if self.path == "/run":
|
| 1843 |
+
mode = str(request.get("mode", "response"))
|
| 1844 |
+
text = str(request.get("text", ""))
|
| 1845 |
+
host_tools = [str(item) for item in request.get("host_tools", []) or []]
|
| 1846 |
+
max_speech_steps = int(request.get("max_speech_steps", args.max_speech_steps))
|
| 1847 |
+
avoid_texts = set(str(item) for item in request.get("avoid_texts", []) or [])
|
| 1848 |
+
canonicalize_stream = bool(request.get("canonicalize_stream", False))
|
| 1849 |
+
if mode == "response":
|
| 1850 |
+
payload = _run_response(
|
| 1851 |
+
artifact,
|
| 1852 |
+
artifact_path,
|
| 1853 |
+
text,
|
| 1854 |
+
host_tools,
|
| 1855 |
+
max_speech_steps,
|
| 1856 |
+
avoid_texts,
|
| 1857 |
+
canonicalize_stream,
|
| 1858 |
+
)
|
| 1859 |
+
elif mode == "history":
|
| 1860 |
+
payload = _run_history(
|
| 1861 |
+
artifact,
|
| 1862 |
+
artifact_path,
|
| 1863 |
+
text,
|
| 1864 |
+
host_tools,
|
| 1865 |
+
max_speech_steps,
|
| 1866 |
+
avoid_texts,
|
| 1867 |
+
canonicalize_stream,
|
| 1868 |
+
)
|
| 1869 |
+
elif mode == "autonomous":
|
| 1870 |
+
payload = _run_autonomous(
|
| 1871 |
+
artifact,
|
| 1872 |
+
artifact_path,
|
| 1873 |
+
text,
|
| 1874 |
+
host_tools,
|
| 1875 |
+
max_speech_steps,
|
| 1876 |
+
avoid_texts,
|
| 1877 |
+
canonicalize_stream,
|
| 1878 |
+
)
|
| 1879 |
+
elif mode == "acoustic":
|
| 1880 |
+
payload = _run_acoustic(
|
| 1881 |
+
artifact,
|
| 1882 |
+
artifact_path,
|
| 1883 |
+
_parse_audio_samples(request.get("audio_samples")),
|
| 1884 |
+
host_tools,
|
| 1885 |
+
max_speech_steps,
|
| 1886 |
+
avoid_texts,
|
| 1887 |
+
float(request.get("audio_symbol_score_floor", 0.25)),
|
| 1888 |
+
)
|
| 1889 |
+
else:
|
| 1890 |
+
raise ValueError("standalone Python Spark service currently supports response, history, autonomous, and acoustic modes")
|
| 1891 |
+
payload["service"] = "sle-artifact-http"
|
| 1892 |
+
body = json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
| 1893 |
+
self.send_response(200 if payload.get("success") else 422)
|
| 1894 |
+
elif self.path == "/v1/chat/completions":
|
| 1895 |
+
payload = _run_openai_chat_completion(
|
| 1896 |
+
artifact,
|
| 1897 |
+
artifact_path,
|
| 1898 |
+
request,
|
| 1899 |
+
int(args.max_speech_steps),
|
| 1900 |
+
)
|
| 1901 |
+
model = str(request.get("model", "saccadic"))
|
| 1902 |
+
if bool(request.get("stream", False)):
|
| 1903 |
+
body = _openai_stream_response(model, payload).encode("utf-8")
|
| 1904 |
+
content_type = "text/event-stream; charset=utf-8"
|
| 1905 |
+
else:
|
| 1906 |
+
response = _openai_completion_response(model, payload)
|
| 1907 |
+
body = json.dumps(response, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
| 1908 |
+
self.send_response(200 if payload.get("success") else 422)
|
| 1909 |
+
else:
|
| 1910 |
+
body = json.dumps({"success": False, "error": "unknown endpoint", "service": "sle-artifact-http"}).encode("utf-8")
|
| 1911 |
+
self.send_response(404)
|
| 1912 |
+
except Exception as exc:
|
| 1913 |
+
body = json.dumps({"success": False, "error": str(exc), "service": "sle-artifact-http"}).encode("utf-8")
|
| 1914 |
+
content_type = "application/json"
|
| 1915 |
+
self.send_response(422)
|
| 1916 |
+
self.send_header("Content-Type", content_type)
|
| 1917 |
+
self.send_header("Content-Length", str(len(body)))
|
| 1918 |
+
self.end_headers()
|
| 1919 |
+
self.wfile.write(body)
|
| 1920 |
+
|
| 1921 |
+
server = HTTPServer((args.host, int(args.port)), Handler)
|
| 1922 |
+
handled = 0
|
| 1923 |
+
while args.max_requests is None or handled < int(args.max_requests):
|
| 1924 |
+
server.handle_request()
|
| 1925 |
+
handled += 1
|
| 1926 |
+
return 0
|
| 1927 |
+
|
| 1928 |
+
|
| 1929 |
+
def _parser() -> argparse.ArgumentParser:
|
| 1930 |
+
parser = argparse.ArgumentParser(prog="sle_spark.py")
|
| 1931 |
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
| 1932 |
+
run = subparsers.add_parser("run-artifact")
|
| 1933 |
+
run.add_argument("--artifact", default="model.cra")
|
| 1934 |
+
run.add_argument("--mode", choices=("response", "history", "autonomous", "acoustic"), default="response")
|
| 1935 |
+
run.add_argument("--text")
|
| 1936 |
+
run.add_argument("--host-tool", action="append", default=[])
|
| 1937 |
+
run.add_argument("--max-speech-steps", type=int, default=18)
|
| 1938 |
+
run.add_argument("--avoid-text", action="append", default=[])
|
| 1939 |
+
run.add_argument("--avoid-replay", action="store_true")
|
| 1940 |
+
run.add_argument("--canonicalize-stream", action="store_true")
|
| 1941 |
+
run.add_argument("--audio-samples")
|
| 1942 |
+
run.add_argument("--audio-symbol-score-floor", type=float, default=0.25)
|
| 1943 |
+
run.add_argument("--turns", type=int, default=1)
|
| 1944 |
+
run.add_argument("--save-artifact", dest="save_artifact")
|
| 1945 |
+
run.add_argument("--output-artifact", dest="save_artifact")
|
| 1946 |
+
run.add_argument("--diverse-speech-limit", type=int)
|
| 1947 |
+
run.add_argument("--punctuation-floor", type=int)
|
| 1948 |
+
run.add_argument("--dt", type=float, default=0.05)
|
| 1949 |
+
run.add_argument("--compose-speech", action="store_true")
|
| 1950 |
+
run.add_argument("--prefer-composed", action="store_true")
|
| 1951 |
+
run.add_argument("--recover-on-failure", action="store_true")
|
| 1952 |
+
run.set_defaults(func=_run_artifact)
|
| 1953 |
+
|
| 1954 |
+
serve = subparsers.add_parser("serve-artifact")
|
| 1955 |
+
serve.add_argument("--artifact", default="model.cra")
|
| 1956 |
+
serve.add_argument("--host", default="127.0.0.1")
|
| 1957 |
+
serve.add_argument("--port", type=int, default=8765)
|
| 1958 |
+
serve.add_argument("--smoke-once", action="store_true")
|
| 1959 |
+
serve.add_argument("--max-requests", type=int)
|
| 1960 |
+
serve.add_argument("--max-speech-steps", type=int, default=18)
|
| 1961 |
+
serve.add_argument("--dt", type=float, default=0.05)
|
| 1962 |
+
serve.set_defaults(func=_serve_artifact)
|
| 1963 |
+
return parser
|
| 1964 |
+
|
| 1965 |
+
|
| 1966 |
+
def main(argv: list[str] | None = None) -> int:
|
| 1967 |
+
args = _parser().parse_args(argv)
|
| 1968 |
+
try:
|
| 1969 |
+
return int(args.func(args))
|
| 1970 |
+
except Exception as exc:
|
| 1971 |
+
print(json.dumps({"success": False, "error": str(exc)}, ensure_ascii=True, sort_keys=True, separators=(",", ":")))
|
| 1972 |
+
return 1
|
| 1973 |
+
|
| 1974 |
+
|
| 1975 |
+
if __name__ == "__main__":
|
| 1976 |
+
raise SystemExit(main())
|
sle_v2_1_release_metrics.json
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"acoustic_stream_success": true,
|
| 3 |
+
"autonomous_action_custom_tool_success": true,
|
| 4 |
+
"conversation_history_success": true,
|
| 5 |
+
"conversation_history_tool_path": [
|
| 6 |
+
"math.add"
|
| 7 |
+
],
|
| 8 |
+
"conversation_history_value": 4,
|
| 9 |
+
"custom_tool_final_value": {
|
| 10 |
+
"customer_id": "CUST-42",
|
| 11 |
+
"renewal_days": 19,
|
| 12 |
+
"tier": "enterprise"
|
| 13 |
+
},
|
| 14 |
+
"custom_tool_path": [
|
| 15 |
+
"crm.lookup_customer"
|
| 16 |
+
],
|
| 17 |
+
"dataset_failure_count": 0,
|
| 18 |
+
"dataset_lane_count": 25,
|
| 19 |
+
"dataset_row_count": 502000,
|
| 20 |
+
"deployment_hosts_success": true,
|
| 21 |
+
"failed_quality_gates": [],
|
| 22 |
+
"failed_release_gates": [],
|
| 23 |
+
"failed_release_smoke_gates": [],
|
| 24 |
+
"general_evidence_success": true,
|
| 25 |
+
"hf_repo_slug": "SLE-V2.1-Omega12K-LX",
|
| 26 |
+
"kind": "sle_v2_1_release_metrics",
|
| 27 |
+
"openai_chat_completion_success": true,
|
| 28 |
+
"openai_streaming_success": true,
|
| 29 |
+
"public_runtime_success": true,
|
| 30 |
+
"readiness_success": true,
|
| 31 |
+
"release_name": "SLE-V2.1-\u03a912K-LX",
|
| 32 |
+
"release_smoke_success": true,
|
| 33 |
+
"system_instruction_success": true,
|
| 34 |
+
"system_instruction_tool_path": [
|
| 35 |
+
"fact.lookup"
|
| 36 |
+
],
|
| 37 |
+
"system_instruction_value": {
|
| 38 |
+
"color": "Blue",
|
| 39 |
+
"response": "Blue stays calm.",
|
| 40 |
+
"system_prompt": "Reply directly without visible thoughts."
|
| 41 |
+
}
|
| 42 |
+
}
|