from __future__ import annotations import argparse import importlib import json import struct import sys import tempfile import time from http.server import BaseHTTPRequestHandler, HTTPServer from pathlib import Path from typing import Any MAGIC = b"SLECRA1\0" TOOL_MAGIC = b"SLETOOL1" COGNITION_MAGIC = b"SLECOG1\0" WORLD_MAGIC = b"SLEWRLD1" LANGUAGE_MAGIC = b"SLELANG1" DIALOGUE_MAGIC = b"SLEDIAL1" INITIATIVE_MAGIC = b"SLEINIT1" AUTONOMOUS_GOAL_MAGIC = b"SLEGOAL1" AUTONOMOUS_ROUTE_MAGIC = b"SLEROUT1" FACT_MAGIC = b"SLEFACT1" STREAM_MAGIC = b"SLESTRM1" SPEECH_MAGIC = b"SLESPEK1" ACOUSTIC_MAGIC = b"SLEAUDI1" ACOUSTIC_SYMBOL_MAGIC = b"SLEASMB1" PUNCTUATION = frozenset((".", ",", "!", "?", ";", ":", "<", ">", "\u061f", "\u3002", "\uff01", "\uff1f")) TERMINAL_PUNCTUATION = frozenset((".", "!", "?", "\u061f", "\u3002", "\uff01", "\uff1f")) ROLE_LABELS = frozenset(("system", "developer", "user", "assistant", "tool")) MASK_64 = 0xFFFFFFFFFFFFFFFF FNV_OFFSET = 0xCBF29CE484222325 FNV_PRIME = 0x100000001B3 ACOUSTIC_SIGNATURE_NAMES = ( "length", "first_bin", "last_bin", "mean_bin", "energy_bin", "peak_bin", "crossings", "direction", ) def _read_exact(handle, length: int) -> bytes: payload = handle.read(length) if len(payload) != length: raise ValueError("truncated CRA artifact") return payload def _read_u32(handle) -> int: return struct.unpack(" int: return struct.unpack(" bytes: return _read_exact(handle, _read_u32(handle)) def _skip_block(handle) -> None: _read_block(handle) def _read_json_block(handle) -> dict[str, Any]: payload = _read_block(handle) if not payload: return {} value = json.loads(payload.decode("utf-8")) if not isinstance(value, dict): raise ValueError("CRA JSON extension block must be an object") return value def _load_artifact(path: str | Path) -> dict[str, Any]: artifact: dict[str, Any] = { "metadata": {}, "tools": {}, "cognition": {}, "language": {}, "instructions": {}, "contexts": {}, "response_policy": {}, "initiative": {}, "autonomous_goals": {}, "autonomous_routes": {}, "facts": {}, "stream": {}, "speech": {}, } with Path(path).open("rb") as handle: if _read_exact(handle, len(MAGIC)) != MAGIC: raise ValueError("not an SLE CRA artifact") version = _read_u32(handle) if version != 1: raise ValueError(f"unsupported CRA version: {version}") artifact["version"] = version metadata_payload = _read_block(handle) artifact["metadata"] = json.loads(metadata_payload.decode("utf-8")) if metadata_payload else {} dimensions = _read_u32(handle) artifact["memory_dimensions"] = dimensions artifact["memory_seed"] = _read_u64(handle) record_count = _read_u32(handle) artifact["memory_records"] = record_count for _ in range(record_count): _skip_block(handle) _skip_block(handle) artifact["liquid_neurons"] = _read_u32(handle) artifact["liquid_inputs"] = _read_u32(handle) artifact["liquid_seed"] = _read_u64(handle) for _ in range(5): _skip_block(handle) marker = handle.read(len(TOOL_MAGIC)) if not marker: return artifact if marker != TOOL_MAGIC: raise ValueError("unknown CRA extension section") tools: dict[str, Any] = {} for _ in range(_read_u32(handle)): name = _read_block(handle).decode("utf-8") payload = json.loads(_read_block(handle).decode("utf-8")) _skip_block(handle) has_avoidance = bool(struct.unpack(" bool: return len(symbol) == 1 and (symbol in PUNCTUATION or not symbol.isalnum()) def _symbol_class(symbol: str) -> str: if len(symbol) == 1 and symbol.isdecimal(): return "decimal" return "text" def _segment_text(text: str) -> list[str]: symbols: list[str] = [] current: list[str] = [] current_class = "" def finish_current() -> None: nonlocal current, current_class if current: symbols.append("".join(current)) current = [] current_class = "" for character in text: if character.isspace(): finish_current() elif _is_punctuation(character): finish_current() symbols.append(character) else: character_class = _symbol_class(character) if current and character_class != current_class: finish_current() current.append(character) current_class = character_class finish_current() return symbols def _render_text(symbols: list[str]) -> str: text = "" in_angle_tag = False angle_tag_content = "" for symbol in symbols: if symbol == "<": text = text.rstrip() + "<" in_angle_tag = True angle_tag_content = "" elif in_angle_tag: if symbol == ">": text = text.rstrip() + ">" if angle_tag_content.startswith("/"): text += " " in_angle_tag = False else: text += symbol angle_tag_content += symbol elif _is_punctuation(symbol): text = text.rstrip() + symbol + " " else: text += symbol + " " return text.rstrip() def _render_slot_value(symbols: list[str], example: Any) -> str: example_text = str(example).strip() if example_text and " " not in example_text: return "".join(symbols) return _render_text(symbols) def _edit_distance(left: str, right: str) -> int: if left == right: return 0 previous = list(range(len(right) + 1)) for index, left_ch in enumerate(left, start=1): current = [index] for other_index, right_ch in enumerate(right, start=1): cost = 0 if left_ch == right_ch else 1 current.append(min(previous[other_index] + 1, current[-1] + 1, previous[other_index - 1] + cost)) previous = current return previous[-1] def _parse_audio_samples(raw: Any) -> list[float]: if raw is None: raise ValueError("acoustic mode requires --audio-samples") if isinstance(raw, str): text = raw.strip() if not text: raise ValueError("audio samples must not be empty") if text.startswith("["): payload = json.loads(text) else: payload = [part.strip() for part in text.split(",") if part.strip()] else: payload = raw if not isinstance(payload, list): raise ValueError("audio samples must be a JSON list or comma-separated numbers") samples = [float(value) for value in payload] if not samples: raise ValueError("audio samples must not be empty") return samples def _stream_evidence(text: str, stream_payload: dict[str, Any], canonicalize: bool = False) -> dict[str, Any]: raw_chunks = _segment_text(text) chunks = list(raw_chunks) alias_names = [""] * len(chunks) if canonicalize: for index, chunk in enumerate(list(chunks)): for raw in stream_payload.get("aliases", []) or []: observed = str(raw.get("observed", "")) max_distance = int(raw.get("max_distance", 0)) if _edit_distance(chunk, observed) <= max_distance: chunks[index] = str(raw.get("canonical", observed)) alias_names[index] = str(raw.get("name", "")) break learned_surfaces: set[str] = set() for raw in stream_payload.get("chunk_surfaces", []) or []: learned_surfaces.update(str(item) for item in raw.get("chunks", []) or []) boundaries: list[str] = [] errors: list[float] = [] for index, chunk in enumerate(chunks): if index == len(chunks) - 1: boundary = "end" elif _is_punctuation(chunk): boundary = "punctuation" elif index + 1 < len(chunks) and _symbol_class(chunk[-1:]) != _symbol_class(chunks[index + 1][:1]): boundary = "category" else: boundary = "space" boundaries.append(boundary) if chunk in learned_surfaces or _is_punctuation(chunk): errors.append(0.0) elif learned_surfaces: best = min(_edit_distance(chunk, learned) / max(len(chunk), len(learned), 1) for learned in learned_surfaces) errors.append(float(best)) else: errors.append(1.0 / max(len(chunk), 1)) mean_error = sum(errors) / len(errors) if errors else 0.0 return { "chunks": chunks, "raw_chunks": raw_chunks, "alias_names": alias_names, "chunk_boundaries": boundaries, "chunk_mean_errors": errors, "category_boundary_count": sum(1 for boundary in boundaries if boundary == "category"), "mean_error": mean_error, "boundary_impulses": len(boundaries), "steps": len(text), "algorithm": "standalone-predictive-semantic-chunking", } def _acoustic_stream_evidence(samples: list[float], acoustic_payload: dict[str, Any]) -> dict[str, Any]: if not acoustic_payload: raise ValueError("artifact does not contain acoustic stream state") bins = int(acoustic_payload.get("bins", 0)) if bins < 2: raise ValueError("acoustic bins must be at least 2") transitions = [float(value) for value in acoustic_payload.get("transitions", []) or []] if len(transitions) != bins * bins: raise ValueError("acoustic transition table length mismatch") error_floor = float(acoustic_payload.get("error_floor", 0.5)) silence_floor = float(acoustic_payload.get("silence_floor", 0.02)) errors = [0.0] * len(samples) boundaries = [0.0] * len(samples) denominator = max(bins - 1, 1) for index in range(1, len(samples)): previous = _acoustic_quantize(samples[index - 1], bins) current = _acoustic_quantize(samples[index], bins) transition = transitions[previous * bins + current] error = 0.0 if transition <= 0.0: error = abs(current - previous) / denominator boundary = 0.0 if error >= error_floor: boundary = 1.0 elif abs(samples[index]) <= silence_floor and abs(samples[index - 1]) > silence_floor: boundary = 1.0 errors[index] = error boundaries[index] = boundary chunks = [] start = 0 for index in range(1, len(samples)): if boundaries[index] <= 0.0: continue chunks.append(_acoustic_chunk(samples, errors, start, index, _acoustic_boundary_name(samples, errors, index, error_floor, silence_floor))) start = index chunks.append(_acoustic_chunk(samples, errors, start, len(samples), "end")) chunks = [chunk for chunk in chunks if chunk["start"] < chunk["end"]] return { "chunks": chunks, "chunk_spans": [[chunk["start"], chunk["end"]] for chunk in chunks], "chunk_boundaries": [chunk["boundary"] for chunk in chunks], "chunk_mean_errors": [chunk["mean_error"] for chunk in chunks], "mean_error": sum(errors) / len(errors) if errors else 0.0, "boundary_impulses": sum(1 for value in boundaries if value > 0.0), "steps": len(samples), "algorithm": "standalone-predictive-acoustic-chunking", } def _acoustic_chunk(samples: list[float], errors: list[float], start: int, end: int, boundary: str) -> dict[str, Any]: width = max(end - start, 1) mean_error = sum(errors[start:end]) / width if end > start else 0.0 return {"start": int(start), "end": int(end), "mean_error": float(mean_error), "boundary": boundary} def _acoustic_boundary_name(samples: list[float], errors: list[float], index: int, error_floor: float, silence_floor: float) -> str: if errors[index] >= error_floor: return "prediction_error" if index > 0 and abs(samples[index]) <= silence_floor and abs(samples[index - 1]) > silence_floor: return "silence" return "boundary" def _acoustic_quantize(value: float, bins: int) -> int: bounded = max(-1.0, min(1.0, float(value))) scaled = (bounded + 1.0) * 0.5 * (bins - 1) index = int(scaled + 0.5) if index < 0: return 0 if index >= bins: return bins - 1 return index def _decode_acoustic_symbols( artifact: dict[str, Any], samples: list[float], chunks: list[dict[str, Any]], min_score: float, ) -> dict[str, Any]: payload = artifact.get("acoustic_symbols", {}) or {} records = list(payload.get("records", []) or []) if not records: raise ValueError("artifact does not contain acoustic symbol grounding state") bins = int(payload.get("bins", 0)) if bins < 2: raise ValueError("acoustic symbol bins must be at least 2") dimensions = int(artifact.get("memory_dimensions", 0)) seed = int(artifact.get("memory_seed", 0)) if dimensions <= 0: raise ValueError("artifact memory dimensions must be positive") matches = [] for chunk in chunks: if _acoustic_chunk_is_silent(samples, chunk, 0.0): continue query = _hdc_encode(_acoustic_signature(samples, chunk["start"], chunk["end"], bins), dimensions, seed) ranked = [] for raw in records: encoded = _hdc_encode(dict(raw.get("signature", {}) or {}), dimensions, seed) score = _vector_resonance(query, encoded) ranked.append((score, str(raw.get("name", "")), raw)) ranked.sort(key=lambda item: (-item[0], item[1])) score, _name, raw = ranked[0] if score < min_score: raise ValueError(f"acoustic symbol score below floor: {score}") matches.append( { "symbol": str(raw.get("symbol", "")), "record_name": str(raw.get("name", "")), "score": float(score), "chunk_span": [int(chunk["start"]), int(chunk["end"])], "chunk_boundary": str(chunk["boundary"]), } ) return { "symbols": [match["symbol"] for match in matches], "matches": matches, "algorithm": "standalone-hdc-acoustic-symbol-grounding", } def _acoustic_signature(samples: list[float], start: int, end: int, bins: int) -> dict[str, int]: if end <= start: return {name: 0 for name in ACOUSTIC_SIGNATURE_NAMES} first = _acoustic_quantize(samples[start], bins) last = _acoustic_quantize(samples[end - 1], bins) total = 0.0 abs_total = 0.0 peak_abs = 0.0 crossings = 0 previous_sign = 0 for index in range(start, end): value = float(samples[index]) total += value abs_value = abs(value) abs_total += abs_value if abs_value > peak_abs: peak_abs = abs_value sign = 0 if value > 0.0: sign = 1 elif value < 0.0: sign = -1 if sign != 0: if previous_sign != 0 and sign != previous_sign: crossings += 1 previous_sign = sign length = end - start mean = total / length energy = abs_total / length energy_bin = _bounded_bin(energy * (bins - 1), bins) peak_bin = _bounded_bin(peak_abs * (bins - 1), bins) direction = 1 if samples[end - 1] > samples[start]: direction = 2 elif samples[end - 1] < samples[start]: direction = 0 return { "length": int(length), "first_bin": int(first), "last_bin": int(last), "mean_bin": int(_acoustic_quantize(mean, bins)), "energy_bin": int(energy_bin), "peak_bin": int(peak_bin), "crossings": int(crossings), "direction": int(direction), } def _bounded_bin(value: float, bins: int) -> int: index = int(value + 0.5) if index < 0: return 0 if index >= bins: return bins - 1 return index def _acoustic_chunk_is_silent(samples: list[float], chunk: dict[str, Any], silence_floor: float) -> bool: for index in range(int(chunk["start"]), int(chunk["end"])): if abs(samples[index]) > silence_floor: return False return True def _hdc_encode(attributes: dict[str, Any], dimensions: int, seed: int) -> list[float]: if not attributes: raise ValueError("HDC encode requires attributes") accumulator = [0.0] * dimensions for role in sorted(attributes): role_vector = _hdc_symbol(str(role), dimensions, seed) value_vector = _hdc_symbol(str(attributes[role]), dimensions, seed) for index in range(dimensions): accumulator[index] += role_vector[index] * value_vector[index] out = [0.0] * dimensions for index, value in enumerate(accumulator): if value > 0.0: out[index] = 1.0 elif value < 0.0: out[index] = -1.0 elif index % 2 == 0: out[index] = 1.0 else: out[index] = -1.0 return out def _hdc_symbol(name: str, dimensions: int, seed: int) -> list[float]: base_hash = _stable_text_hash(name, seed) seed_u64 = int(seed) & MASK_64 out = [0.0] * dimensions for index in range(dimensions): mixed = _mix64(base_hash ^ seed_u64 ^ index) out[index] = 1.0 if mixed & 1 else -1.0 return out def _stable_text_hash(text: str, seed: int) -> int: value = (FNV_OFFSET ^ (int(seed) & MASK_64)) & MASK_64 for byte in text.encode("utf-8"): value ^= byte value = (value * FNV_PRIME) & MASK_64 return value def _mix64(value: int) -> int: z = (int(value) + 0x9E3779B97F4A7C15) & MASK_64 z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & MASK_64 z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & MASK_64 return (z ^ (z >> 31)) & MASK_64 def _vector_resonance(left: list[float], right: list[float]) -> float: if len(left) != len(right) or not left: raise ValueError("resonance vectors must have the same positive length") total = 0.0 for left_value, right_value in zip(left, right): total += left_value * right_value return total / len(left) def _match_pattern( pattern: list[str], symbols: list[str], examples: dict[str, Any], min_literal_score: float = 1.0, ) -> tuple[dict[str, Any], float] | None: slots: dict[str, Any] = {} literal_scores: list[float] = [] pattern_index = 0 symbol_index = 0 while pattern_index < len(pattern): if symbol_index >= len(symbols): return None expected = pattern[pattern_index] if expected.startswith("$"): key = expected[1:] example = examples.get(key, "") slot_width = max(1, len(_segment_text(str(example)))) if symbol_index + slot_width > len(symbols): return None observed_symbols = symbols[symbol_index:symbol_index + slot_width] try: slots[key] = _coerce_like(example, _render_slot_value(observed_symbols, example)) except (TypeError, ValueError): return None symbol_index += slot_width else: observed = symbols[symbol_index] width = max(len(expected), len(observed), 1) score = 1.0 - (_edit_distance(expected, observed) / width) if score < min_literal_score: return None literal_scores.append(score) symbol_index += 1 pattern_index += 1 if symbol_index != len(symbols): return None score = sum(literal_scores) / len(literal_scores) if literal_scores else 1.0 return slots, score def _goal_from_instruction(raw: dict[str, Any], slots: dict[str, Any]) -> dict[str, Any]: goal = {} for key, value in dict(raw.get("goal_template", {}) or {}).items(): if isinstance(value, str) and value.startswith("$"): goal[str(key)] = slots[value[1:]] else: goal[str(key)] = value return goal def _coerce_like(example: Any, value: str) -> Any: if isinstance(example, bool): return value == "True" if isinstance(example, int) and not isinstance(example, bool): return int(value) if isinstance(example, float): return float(value) return value def _parse_instruction(artifact: dict[str, Any], symbols: list[str]) -> dict[str, Any]: candidates: list[tuple[float, str, dict[str, Any]]] = [] for raw in artifact.get("instructions", {}).get("records", []) or []: match = _match_pattern( [str(item) for item in raw.get("pattern", [])], symbols, dict(raw.get("slot_examples", {}) or {}), float(raw.get("min_literal_score", 1.0)), ) if match is None: continue slots, score = match goal = _goal_from_instruction(raw, slots) candidates.append((score, str(raw.get("name", "")), goal)) if not candidates: raise ValueError("no instruction pattern matched") candidates.sort(key=lambda item: (-item[0], item[1])) score, name, goal = candidates[0] return {"record_name": name, "goal": goal, "score": score} def _system_context_priority(symbols: list[str], goal: dict[str, Any]) -> int: if len(symbols) < 4: return 0 has_system_role = symbols[0] == "System" and symbols[1] == ":" has_user_role = any( symbols[index] == "User" and symbols[index + 1] == ":" for index in range(len(symbols) - 1) ) has_system_goal = ( "system_prompt" in goal or str(goal.get("query_kind", "")) == "system_instruction" ) return 1 if has_system_role and has_user_role and has_system_goal else 0 def _parse_instruction_suffix(artifact: dict[str, Any], symbols: list[str]) -> dict[str, Any]: candidates: list[tuple[float, int, int, int, str, dict[str, Any], list[str]]] = [] records = artifact.get("instructions", {}).get("records", []) or [] for suffix_start in range(len(symbols)): candidate_symbols = symbols[suffix_start:] if not candidate_symbols or _is_punctuation(candidate_symbols[0]): continue for raw in records: pattern = [str(item) for item in raw.get("pattern", [])] match = _match_pattern( pattern, candidate_symbols, dict(raw.get("slot_examples", {}) or {}), float(raw.get("min_literal_score", 1.0)), ) if match is None: continue slots, score = match goal = _goal_from_instruction(raw, slots) candidates.append( ( score, _system_context_priority(candidate_symbols, goal), suffix_start, len(pattern), str(raw.get("name", "")), goal, candidate_symbols, ) ) if not candidates: raise ValueError("no instruction pattern matched") candidates.sort(key=lambda item: (-item[0], -item[1], -item[2], -item[3], item[4])) score, _system_priority, suffix_start, _pattern_length, name, goal, selected_symbols = candidates[0] return { "record_name": name, "goal": goal, "score": score, "selected_suffix_start": suffix_start, "selected_symbols": selected_symbols, } def _parse_context(artifact: dict[str, Any], symbols: list[str]) -> dict[str, Any]: candidates: list[tuple[float, str, dict[str, Any]]] = [] for raw in artifact.get("contexts", {}).get("records", []) or []: match = _match_pattern( [str(item) for item in raw.get("pattern", [])], symbols, dict(raw.get("slot_examples", {}) or {}), float(raw.get("min_literal_score", 1.0)), ) if match is None: continue slots, score = match context = {} for key, value in dict(raw.get("context_template", {}) or {}).items(): if isinstance(value, str) and value.startswith("$"): context[str(key)] = slots[value[1:]] else: context[str(key)] = value candidates.append((score, str(raw.get("name", "")), context)) if not candidates: raise ValueError("no context pattern matched") candidates.sort(key=lambda item: (-item[0], item[1])) score, name, context = candidates[0] return {"record_name": name, "context": context, "score": score} def _compatible(learned: dict[str, Any], observed: dict[str, Any]) -> bool: return all(key not in observed or observed[key] == value for key, value in learned.items()) def _overlap_score(learned: dict[str, Any], observed: dict[str, Any]) -> tuple[int, int]: overlap = sum(1 for key, value in learned.items() if key in observed and observed[key] == value) mismatch = sum(1 for key, value in learned.items() if key in observed and observed[key] != value) return overlap, -mismatch def _resolve_path(path: str, memory: dict[str, Any]) -> Any: value: Any = memory for part in path.split("."): if isinstance(value, dict): value = value[part] elif isinstance(value, list): value = value[int(part)] else: raise ValueError(f"missing working memory value: {path}") return value def _load_host_tool(spec: str): if "=" not in spec: if spec == "math.add": return spec, lambda a, b: int(a) + int(b) if spec == "math.power": return spec, lambda base, exponent: int(base) ** int(exponent) raise ValueError("custom host tool must use ARTIFACT_TOOL=module:function") tool_name, target = (part.strip() for part in spec.split("=", 1)) module_name, attr_path = (part.strip() for part in target.split(":", 1)) cwd = str(Path.cwd()) if cwd not in sys.path: sys.path.insert(0, cwd) value: Any = importlib.import_module(module_name) for part in attr_path.split("."): value = getattr(value, part) if not callable(value): raise ValueError(f"custom host tool target is not callable: {target}") return tool_name, value def _builtin_host_tools() -> dict[str, Any]: return { "math.add": lambda a, b: int(a) + int(b), "math.power": lambda base, exponent: int(base) ** int(exponent), } def _host_tools(specs: list[str]) -> dict[str, Any]: tools = _builtin_host_tools() for spec in specs: name, func = _load_host_tool(spec) tools[name] = func return tools def _select_procedure(artifact: dict[str, Any], goal: dict[str, Any]) -> dict[str, Any]: candidates = [] for raw in artifact.get("cognition", {}).get("procedures", []) or []: learned = dict(raw.get("goal", {}) or {}) if not _compatible(learned, goal): continue candidates.append((_overlap_score(learned, goal), str(raw.get("name", "")), raw)) if not candidates: raise ValueError("no procedures recorded") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1])) return dict(candidates[0][2]) def _select_route(artifact: dict[str, Any], context: dict[str, Any]) -> dict[str, Any]: candidates = [] for raw in artifact.get("autonomous_routes", {}).get("records", []) or []: learned = dict(raw.get("context", {}) or {}) if not _compatible(learned, context): continue candidates.append((_overlap_score(learned, context), int(raw.get("uses", 0)), str(raw.get("name", "")), raw)) if not candidates: raise ValueError("context does not contain learned autonomous route keys") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2])) raw = dict(candidates[0][3]) return { "record_name": str(raw.get("name", "")), "route": str(raw.get("route", "")), "score": float(candidates[0][0][0]), } def _select_autonomous_goal(artifact: dict[str, Any], context: dict[str, Any]) -> dict[str, Any]: candidates = [] for raw in artifact.get("autonomous_goals", {}).get("records", []) or []: learned = dict(raw.get("context", {}) or {}) if not _compatible(learned, context): continue candidates.append((_overlap_score(learned, context), int(raw.get("uses", 0)), str(raw.get("name", "")), raw)) if not candidates: raise ValueError("context does not contain learned autonomous goal keys") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2])) raw = dict(candidates[0][3]) goal = dict(raw.get("goal", {}) or {}) transfers = [] for source, target in dict(raw.get("goal_from_context", {}) or {}).items(): if source in context: goal[str(target)] = context[source] transfers.append([str(source), str(target), str(context[source])]) return { "record_name": str(raw.get("name", "")), "goal": goal, "score": float(candidates[0][0][0]), "context_transfers": transfers, } def _select_initiative(artifact: dict[str, Any], context: dict[str, Any]) -> dict[str, Any]: candidates = [] for raw in artifact.get("initiative", {}).get("records", []) or []: learned = dict(raw.get("context", {}) or {}) if not _compatible(learned, context): continue candidates.append((_overlap_score(learned, context), int(raw.get("uses", 0)), str(raw.get("name", "")), raw)) if not candidates: raise ValueError("context does not contain learned initiative keys") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2])) raw = dict(candidates[0][3]) meaning = dict(raw.get("speech_meaning", {}) or {}) transfers = [] for source, target in dict(raw.get("meaning_from_context", {}) or {}).items(): if source not in context: raise ValueError(f"initiative context missing meaning key: {source}") meaning[str(target)] = context[source] transfers.append([str(source), str(target), str(context[source])]) return { "record_name": str(raw.get("name", "")), "meaning": meaning, "score": float(candidates[0][0][0]), "context_transfers": transfers, } def _query_facts(artifact: dict[str, Any], arguments: dict[str, Any]) -> dict[str, Any]: candidates = [] for raw in artifact.get("facts", {}).get("records", []) or []: attrs = dict(raw.get("attributes", {}) or {}) score = _overlap_score(attrs, arguments) if score[0] > 0: candidates.append((score, str(raw.get("name", "")), attrs)) if not candidates: raise ValueError("no facts matched") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1])) return candidates[0][2] def _solve(artifact: dict[str, Any], goal: dict[str, Any], tools: dict[str, Any]) -> dict[str, Any]: procedure = _select_procedure(artifact, goal) working = dict(goal) steps_out = [] for raw_step in procedure.get("steps", []) or []: template = dict(raw_step.get("arguments", {}) or {}) arguments = {} for key, value in template.items(): if isinstance(value, str) and value.startswith("$"): arguments[str(key)] = _resolve_path(value[1:], working) else: arguments[str(key)] = value tool_name = str(raw_step.get("tool_name", "")) if tool_name == "fact.lookup": result = _query_facts(artifact, arguments) else: func = tools.get(tool_name) if func is None: raise ValueError(f"tool not available: {tool_name}") result = func(**arguments) working[str(raw_step.get("output", ""))] = result steps_out.append({"tool_name": tool_name, "arguments": arguments, "value": result}) final_value = steps_out[-1]["value"] if steps_out else None return { "procedure_name": str(procedure.get("name", "")), "success": True, "steps": steps_out, "working_memory": working, "final_value": final_value, } def _response_decision(artifact: dict[str, Any], goal: dict[str, Any], final_value: Any) -> dict[str, Any]: candidates = [] for raw in artifact.get("response_policy", {}).get("records", []) or []: learned = dict(raw.get("goal", {}) or {}) if not _compatible(learned, goal): continue candidates.append((_overlap_score(learned, goal), str(raw.get("name", "")), raw)) if not candidates: raise ValueError("goal does not match any learned response policy") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1])) raw = dict(candidates[0][2]) meaning = dict(raw.get("response_meaning", {}) or {}) response_slot = str(raw.get("response_slot", "")) if isinstance(final_value, dict): meaning.update(final_value) meaning[response_slot] = final_value else: meaning[response_slot] = final_value for source, target in dict(raw.get("meaning_from_goal", {}) or {}).items(): if source in goal: meaning[str(target)] = goal[source] return {"record_name": str(raw.get("name", "")), "meaning": meaning, "response_slot": response_slot} def _generate_speech(artifact: dict[str, Any], meaning: dict[str, Any], max_steps: int, avoid_texts: set[str]) -> dict[str, Any]: records = list(artifact.get("speech", {}).get("records", []) or []) learned_texts = set(str(text) for text in artifact.get("speech", {}).get("learned_texts", []) or []) previous = "" trajectory = "" trajectory_index = -1 symbols: list[str] = [] transition_names: list[str] = [] for _ in range(max_steps): candidates = [] for raw in records: if str(raw.get("previous", "")) != previous: continue learned = dict(raw.get("meaning", {}) or {}) if not _compatible(learned, meaning): continue record_trajectory = _transition_trajectory_name(raw) record_index = _transition_index(raw) if trajectory and record_trajectory != trajectory: continue if record_index >= 0 and record_index != trajectory_index + 1: continue candidates.append( ( _overlap_score(learned, meaning), _speech_trajectory_uses(records, record_trajectory), int(raw.get("uses", 0)), str(raw.get("name", "")), raw, ) ) if not candidates: raise ValueError(f"no speech transition learned after: {previous}") candidates.sort(key=lambda item: (-item[0][0], item[0][1], item[1], item[2], item[3])) raw = dict(candidates[0][4]) if not trajectory: trajectory = _transition_trajectory_name(raw) trajectory_index = _transition_index(raw) transition_names.append(str(raw.get("name", ""))) symbol = str(raw.get("symbol", "")) if symbol == "": text = _render_text(symbols) emitted = [text] if text in learned_texts or text in avoid_texts else [] return { "record_name": transition_names[0] if transition_names else "", "symbols": symbols, "transition_names": transition_names, "score": float(len(transition_names)), "text": text, "algorithm": "standalone-hdc-transition-speech", "composed": False, "trajectory_names": _speech_trajectory_names(records, transition_names), "punctuation_symbols": [symbol for symbol in symbols if _is_punctuation(symbol)], "emoji_symbols": [], "slot_transfers": _slot_transfers(records, transition_names, meaning), "replay_evidence": [], "emitted_replay_evidence": emitted, "blocked_replay_evidence": [], } emitted_symbol = str(meaning[symbol[1:]]) if symbol.startswith("$") else symbol symbols.append(emitted_symbol) previous = symbol raise ValueError("speech rollout did not reach an end transition") def _transition_trajectory_name(raw: dict[str, Any]) -> str: name = str(raw.get("name", "")) if ":" not in name: return name return name.rsplit(":", 1)[0] def _transition_index(raw: dict[str, Any]) -> int: name = str(raw.get("name", "")) if ":" not in name: return -1 try: return int(name.rsplit(":", 1)[1]) except ValueError: return -1 def _speech_trajectory_names(records: list[dict[str, Any]], transition_names: list[str]) -> list[str]: by_name = {str(raw.get("name", "")): raw for raw in records} names = [] seen = set() for name in transition_names: raw = by_name.get(name) if raw is None or str(raw.get("symbol", "")) == "": continue trajectory = _transition_trajectory_name(raw) if trajectory and trajectory not in seen: seen.add(trajectory) names.append(trajectory) return names def _speech_trajectory_uses(records: list[dict[str, Any]], trajectory: str) -> int: total = 0 for raw in records: if _transition_trajectory_name(raw) == trajectory: total += int(raw.get("uses", 0)) return total def _slot_transfers(records: list[dict[str, Any]], transition_names: list[str], meaning: dict[str, Any]) -> list[list[str]]: by_name = {str(raw.get("name", "")): raw for raw in records} transfers = [] seen = set() for name in transition_names: symbol = str(by_name.get(name, {}).get("symbol", "")) if not symbol.startswith("$"): continue key = symbol[1:] if key in meaning and key not in seen: transfers.append([key, str(meaning[key])]) seen.add(key) return transfers def _speech_transition_use_counts(records: list[dict[str, Any]], transition_names: list[str]) -> list[list[Any]]: by_name = {str(raw.get("name", "")): raw for raw in records} counts = [] for name in transition_names: raw = by_name.get(name) if raw is not None: counts.append([name, int(raw.get("uses", 0))]) return counts def _record_speech_usage(speech_payload: dict[str, Any], transition_names: list[str]) -> None: selected = set(str(name) for name in transition_names) for raw in speech_payload.get("records", []) or []: if str(raw.get("name", "")) in selected: raw["uses"] = int(raw.get("uses", 0)) + 1 def _copy_exact(handle, out, length: int) -> bytes: payload = _read_exact(handle, length) out.write(payload) return payload def _copy_block(handle, out) -> bytes: length_payload = _read_exact(handle, 4) length = struct.unpack(" None: out.write(struct.pack(" None: target = Path(target_path) target.parent.mkdir(parents=True, exist_ok=True) with Path(source_path).open("rb") as handle, target.open("wb") as out: if _copy_exact(handle, out, len(MAGIC)) != MAGIC: raise ValueError("not an SLE CRA artifact") _copy_exact(handle, out, 4) _copy_block(handle, out) memory_header = _copy_exact(handle, out, 4 + 8 + 4) record_count = struct.unpack(" dict[str, Any]: tools = _host_tools(host_tool_specs) stream = _stream_evidence(text, artifact.get("stream", {}), canonicalize=canonicalize_stream) instruction = _parse_instruction_suffix(artifact, [str(chunk) for chunk in stream["chunks"]]) cognition = _solve(artifact, dict(instruction["goal"]), tools) response = _response_decision(artifact, dict(instruction["goal"]), cognition["final_value"]) speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts) steps = cognition["steps"] arguments = steps[0]["arguments"] if steps else {} tool_path = [step["tool_name"] for step in steps] emitted_replay = speech["emitted_replay_evidence"] return { "success": not bool(emitted_replay), "mode": "response", "artifact": artifact_path, "host_tools": host_tool_specs, "instruction_record": instruction["record_name"], "instruction_goal": instruction["goal"], "selected_suffix_start": instruction["selected_suffix_start"], "selected_symbols": instruction["selected_symbols"], "tool_path": tool_path, "arguments": arguments, "final_value": cognition["final_value"], "text": speech["text"], "algorithm": speech["algorithm"], "transition_names": speech["transition_names"], "punctuation_symbols": speech["punctuation_symbols"], "emoji_symbols": speech["emoji_symbols"], "slot_transfers": speech["slot_transfers"], "replay_evidence": speech["replay_evidence"], "emitted_replay_evidence": emitted_replay, "blocked_replay_evidence": speech["blocked_replay_evidence"], "stream": stream, } def _run_autonomous( artifact: dict[str, Any], artifact_path: str, text: str, host_tool_specs: list[str], max_speech_steps: int, avoid_texts: set[str], canonicalize_stream: bool, ) -> dict[str, Any]: tools = _host_tools(host_tool_specs) stream = _stream_evidence(text, artifact.get("stream", {}), canonicalize=canonicalize_stream) context = _parse_context(artifact, [str(chunk) for chunk in stream["chunks"]]) route = _select_route(artifact, dict(context["context"])) if route["route"] == "speech": initiative = _select_initiative(artifact, dict(context["context"])) speech = _generate_speech(artifact, dict(initiative["meaning"]), max_speech_steps, avoid_texts) emitted_replay = speech["emitted_replay_evidence"] return { "success": not bool(emitted_replay), "mode": "autonomous", "artifact": artifact_path, "host_tools": host_tool_specs, "route_record": route["record_name"], "route": route["route"], "context_record": context["record_name"], "context": context["context"], "context_score": context["score"], "decision": initiative["record_name"], "speech_meaning": initiative["meaning"], "initiative_context_transfers": initiative["context_transfers"], "goal_context_transfers": [], "tool_path": [], "arguments": [], "final_value": None, "text": speech["text"], "algorithm": speech["algorithm"], "transition_names": speech["transition_names"], "punctuation_symbols": speech["punctuation_symbols"], "emoji_symbols": speech["emoji_symbols"], "slot_transfers": speech["slot_transfers"], "replay_evidence": speech["replay_evidence"], "emitted_replay_evidence": emitted_replay, "blocked_replay_evidence": speech["blocked_replay_evidence"], "stream": stream, } if route["route"] != "action": raise ValueError(f"unsupported standalone autonomous route: {route['route']}") goal_decision = _select_autonomous_goal(artifact, dict(context["context"])) cognition = _solve(artifact, dict(goal_decision["goal"]), tools) response = _response_decision(artifact, dict(goal_decision["goal"]), cognition["final_value"]) speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts) steps = cognition["steps"] emitted_replay = speech["emitted_replay_evidence"] return { "success": not bool(emitted_replay), "mode": "autonomous", "artifact": artifact_path, "host_tools": host_tool_specs, "route_record": route["record_name"], "route": route["route"], "context_record": context["record_name"], "context": context["context"], "context_score": context["score"], "decision": goal_decision["record_name"], "goal_context_transfers": goal_decision["context_transfers"], "tool_path": [step["tool_name"] for step in steps], "arguments": [step["arguments"] for step in steps], "final_value": cognition["final_value"], "text": speech["text"], "algorithm": speech["algorithm"], "transition_names": speech["transition_names"], "punctuation_symbols": speech["punctuation_symbols"], "emoji_symbols": speech["emoji_symbols"], "slot_transfers": speech["slot_transfers"], "replay_evidence": speech["replay_evidence"], "emitted_replay_evidence": emitted_replay, "blocked_replay_evidence": speech["blocked_replay_evidence"], "stream": stream, } def _run_autonomous_turns( artifact_path: str, text: str, host_tool_specs: list[str], max_speech_steps: int, avoid_texts: set[str], canonicalize_stream: bool, turns: int, save_artifact: str | None, diverse_speech_limit: int | None, punctuation_floor: int | None, ) -> dict[str, Any]: if turns < 1: raise ValueError("--turns must be at least 1") if diverse_speech_limit is not None and diverse_speech_limit < 1: raise ValueError("--diverse-speech-limit must be positive") if punctuation_floor is not None and punctuation_floor < 0: raise ValueError("--punctuation-floor must be non-negative") turn_records = [] with tempfile.TemporaryDirectory() as tmp: current_path = Path(artifact_path) for turn_index in range(turns): artifact = _load_artifact(current_path) payload = _run_autonomous( artifact, str(current_path), text, host_tool_specs, max_speech_steps, avoid_texts, canonicalize_stream, ) records = list(artifact.get("speech", {}).get("records", []) or []) before = _speech_transition_use_counts(records, payload.get("transition_names", [])) _record_speech_usage(artifact.get("speech", {}), payload.get("transition_names", [])) after = _speech_transition_use_counts( list(artifact.get("speech", {}).get("records", []) or []), payload.get("transition_names", []), ) payload["turn_index"] = turn_index payload["transition_use_counts_before"] = before payload["transition_use_counts_after"] = after turn_records.append(payload) if turn_index == turns - 1 and save_artifact: next_path = Path(save_artifact) else: next_path = Path(tmp) / f"turn_{turn_index}.cra" _write_artifact_with_speech_payload(current_path, next_path, artifact.get("speech", {})) current_path = next_path texts = [str(turn.get("text", "")) for turn in turn_records] emitted_replay = [] blocked_replay = [] punctuation = [] for turn in turn_records: emitted_replay.extend(turn.get("emitted_replay_evidence", []) or []) blocked_replay.extend(turn.get("blocked_replay_evidence", []) or []) punctuation.extend(str(symbol) for symbol in turn.get("punctuation_symbols", []) or []) distinct_punctuation = sorted(set(punctuation)) punctuation_success = punctuation_floor is None or len(distinct_punctuation) >= punctuation_floor return { "success": all(bool(turn.get("success")) for turn in turn_records) and not bool(emitted_replay) and punctuation_success, "mode": "autonomous", "artifact": artifact_path, "host_tools": host_tool_specs, "turn_count": len(turn_records), "write_reload": turns > 1, "saved_artifact": str(save_artifact or ""), "diverse_speech_limit": diverse_speech_limit, "punctuation_floor": punctuation_floor, "texts": texts, "text": texts[-1] if texts else "", "unique_text_count": len(set(texts)), "distinct_punctuation_symbols": distinct_punctuation, "punctuation_success": punctuation_success, "turn_records": turn_records, "emitted_replay_evidence": emitted_replay, "blocked_replay_evidence": blocked_replay, } def _instruction_segments(symbols: list[str]) -> list[list[str]]: segments: list[list[str]] = [] current: list[str] = [] for symbol in symbols: current.append(symbol) if ( symbol == ":" and len(current) == 2 and str(current[0]).lower() in ROLE_LABELS ): segments.append(current) current = [] elif symbol in TERMINAL_PUNCTUATION: segments.append(current) current = [] if current: segments.append(current) return [segment for segment in segments if segment] def _history_instruction_segments(symbols: list[str]) -> list[list[str]]: role_segments: list[list[str]] = [] current: list[str] = [] saw_role = False index = 0 while index < len(symbols): symbol = symbols[index] if ( index + 1 < len(symbols) and str(symbol).lower() in ROLE_LABELS and symbols[index + 1] == ":" ): if current: role_segments.append(current) current = [symbol, symbols[index + 1]] saw_role = True index += 2 continue current.append(symbol) index += 1 if current: role_segments.append(current) if saw_role: return [segment for segment in role_segments if segment] raw_segments = _instruction_segments(symbols) segments: list[list[str]] = [] index = 0 while index < len(raw_segments): segment = raw_segments[index] if segment and segment[-1] == ":" and index + 1 < len(raw_segments): segments.append(segment + raw_segments[index + 1]) index += 2 else: segments.append(segment) index += 1 return segments def _segment_role(segment: list[str]) -> str: if len(segment) >= 2 and segment[1] == ":": role = str(segment[0]).lower() if role in ROLE_LABELS: return role return "" def _history_search_order(segments: list[list[str]]) -> list[int]: user_indices = [index for index, segment in enumerate(segments) if _segment_role(segment) == "user"] if user_indices: return list(reversed(user_indices)) non_assistant_indices = [ index for index, segment in enumerate(segments) if _segment_role(segment) != "assistant" ] if non_assistant_indices: return list(reversed(non_assistant_indices)) return list(range(len(segments) - 1, -1, -1)) def _history_candidate_segments(segments: list[list[str]], segment_index: int) -> list[list[str]]: segment = segments[segment_index] candidates: list[list[str]] = [] if _segment_role(segment) == "user": prefix: list[str] = [] for prior_index in range(segment_index - 1, -1, -1): prior = segments[prior_index] role = _segment_role(prior) if role in ("system", "developer"): prefix = prior + prefix continue if role in ("assistant", "user", "tool"): break if prefix: candidates.append(prefix + segment) candidates.append(segment) return candidates def _run_history( artifact: dict[str, Any], artifact_path: str, text: str, host_tool_specs: list[str], max_speech_steps: int, avoid_texts: set[str], canonicalize_stream: bool, ) -> dict[str, Any]: tools = _host_tools(host_tool_specs) stream = _stream_evidence(text, artifact.get("stream", {}), canonicalize=canonicalize_stream) segments = _history_instruction_segments([str(chunk) for chunk in stream["chunks"]]) parse_errors: list[str] = [] for segment_index in _history_search_order(segments): for segment in _history_candidate_segments(segments, segment_index): for suffix_start in range(len(segment)): candidate = segment[suffix_start:] if not candidate or _is_punctuation(candidate[0]): continue try: instruction = _parse_instruction(artifact, candidate) cognition = _solve(artifact, dict(instruction["goal"]), tools) response = _response_decision(artifact, dict(instruction["goal"]), cognition["final_value"]) speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts) except ValueError as exc: parse_errors.append(str(exc)) continue steps = cognition["steps"] emitted_replay = speech["emitted_replay_evidence"] return { "success": not bool(emitted_replay), "mode": "history", "artifact": artifact_path, "host_tools": host_tool_specs, "history_segments": segments, "selected_segment_index": segment_index, "selected_suffix_start": suffix_start, "selected_symbols": candidate, "instruction_record": instruction["record_name"], "instruction_goal": instruction["goal"], "tool_path": [step["tool_name"] for step in steps], "arguments": [step["arguments"] for step in steps], "final_value": cognition["final_value"], "text": speech["text"], "algorithm": speech["algorithm"], "transition_names": speech["transition_names"], "punctuation_symbols": speech["punctuation_symbols"], "emoji_symbols": speech["emoji_symbols"], "slot_transfers": speech["slot_transfers"], "replay_evidence": speech["replay_evidence"], "emitted_replay_evidence": emitted_replay, "blocked_replay_evidence": speech["blocked_replay_evidence"], "stream": stream, } error = "no parseable instruction found in conversation history" if parse_errors: error = f"{error}: {parse_errors[-1]}" return { "success": False, "mode": "history", "artifact": artifact_path, "host_tools": host_tool_specs, "history_segments": segments, "selected_segment_index": -1, "selected_suffix_start": -1, "selected_symbols": [], "stream": stream, "error": error, } def _run_acoustic( artifact: dict[str, Any], artifact_path: str, audio_samples: list[float], host_tool_specs: list[str], max_speech_steps: int, avoid_texts: set[str], min_symbol_score: float, ) -> dict[str, Any]: tools = _host_tools(host_tool_specs) stream = _acoustic_stream_evidence(audio_samples, artifact.get("acoustic", {}) or {}) symbol_run = _decode_acoustic_symbols( artifact, audio_samples, list(stream["chunks"]), float(min_symbol_score), ) instruction = _parse_instruction(artifact, [str(symbol) for symbol in symbol_run["symbols"]]) cognition = _solve(artifact, dict(instruction["goal"]), tools) response = _response_decision(artifact, dict(instruction["goal"]), cognition["final_value"]) speech = _generate_speech(artifact, dict(response["meaning"]), max_speech_steps, avoid_texts) steps = cognition["steps"] emitted_replay = speech["emitted_replay_evidence"] return { "success": not bool(emitted_replay), "mode": "acoustic", "artifact": artifact_path, "host_tools": host_tool_specs, "symbols": symbol_run["symbols"], "symbol_algorithm": symbol_run["algorithm"], "symbol_matches": symbol_run["matches"], "instruction_record": instruction["record_name"], "instruction_goal": instruction["goal"], "tool_path": [step["tool_name"] for step in steps], "arguments": [step["arguments"] for step in steps], "final_value": cognition["final_value"], "text": speech["text"], "texts": [speech["text"]], "algorithm": speech["algorithm"], "transition_names": speech["transition_names"], "punctuation_symbols": speech["punctuation_symbols"], "emoji_symbols": speech["emoji_symbols"], "slot_transfers": speech["slot_transfers"], "replay_evidence": speech["replay_evidence"], "emitted_replay_evidence": emitted_replay, "blocked_replay_evidence": speech["blocked_replay_evidence"], "stream": stream, } def _openai_message_content(content: Any) -> str: if isinstance(content, str): return content if isinstance(content, list): parts: list[str] = [] for item in content: if isinstance(item, str): parts.append(item) elif isinstance(item, dict): if "text" in item: parts.append(str(item["text"])) elif "content" in item: parts.append(str(item["content"])) return " ".join(part for part in parts if part) if content is None: return "" return str(content) def _openai_role_label(role: Any) -> str: name = str(role or "user").strip().lower() if name not in ROLE_LABELS: name = "user" return name.capitalize() def _openai_messages_to_history_text(messages: Any) -> str: if not isinstance(messages, list): raise ValueError("OpenAI chat completion payload requires messages as a list") segments = [] for raw in messages: if not isinstance(raw, dict): raise ValueError("OpenAI chat message must be an object") content = _openai_message_content(raw.get("content", "")) if not content: continue segments.append(f"{_openai_role_label(raw.get('role', 'user'))}: {content}") if not segments: raise ValueError("OpenAI chat completion payload requires at least one non-empty message") return " ".join(segments) def _extend_tool_specs(specs: list[str], value: Any) -> None: if value is None: return if isinstance(value, str): if value: specs.append(value) return if isinstance(value, list): for item in value: if item: specs.append(str(item)) def _openai_host_tool_specs(request: dict[str, Any]) -> list[str]: specs: list[str] = [] _extend_tool_specs(specs, request.get("host_tools")) _extend_tool_specs(specs, request.get("saccadic_host_tools")) saccadic_options = request.get("saccadic") if isinstance(saccadic_options, dict): _extend_tool_specs(specs, saccadic_options.get("host_tools")) tools = request.get("tools") if isinstance(tools, list): for tool in tools: if not isinstance(tool, dict): continue spec = tool.get("saccadic_host_tool") or tool.get("x_saccadic_host_tool") if spec: specs.append(str(spec)) return specs def _openai_request_option(request: dict[str, Any], key: str, default: Any) -> Any: if key in request: return request[key] saccadic_options = request.get("saccadic") if isinstance(saccadic_options, dict) and key in saccadic_options: return saccadic_options[key] return default def _run_openai_chat_completion( artifact: dict[str, Any], artifact_path: str, request: dict[str, Any], default_max_speech_steps: int, ) -> dict[str, Any]: text = _openai_messages_to_history_text(request.get("messages")) payload = _run_history( artifact, artifact_path, text, _openai_host_tool_specs(request), int(_openai_request_option(request, "max_speech_steps", default_max_speech_steps)), set(str(item) for item in _openai_request_option(request, "avoid_texts", []) or []), bool(_openai_request_option(request, "canonicalize_stream", False)), ) payload["service"] = "sle-artifact-http" return payload def _openai_completion_id(payload: dict[str, Any]) -> str: text = str(payload.get("text", "")) value = _stable_text_hash(text, int(payload.get("memory_seed", 0) or 0)) return f"chatcmpl-saccadic-{value:016x}" def _openai_completion_response(model: str, payload: dict[str, Any]) -> dict[str, Any]: content = str(payload.get("text", "")) return { "id": _openai_completion_id(payload), "object": "chat.completion", "created": int(time.time()), "model": model, "choices": [ { "index": 0, "message": {"role": "assistant", "content": content}, "finish_reason": "stop" if payload.get("success") else "error", } ], "usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}, "saccadic": payload, } def _openai_content_chunks(text: str) -> list[str]: chunks: list[str] = [] current = "" for character in text: current += character if character.isspace(): chunks.append(current) current = "" if current: chunks.append(current) return chunks def _openai_stream_response(model: str, payload: dict[str, Any]) -> str: completion_id = _openai_completion_id(payload) created = int(time.time()) events: list[dict[str, Any]] = [ { "id": completion_id, "object": "chat.completion.chunk", "created": created, "model": model, "choices": [{"index": 0, "delta": {"role": "assistant"}, "finish_reason": None}], } ] for chunk in _openai_content_chunks(str(payload.get("text", ""))): events.append( { "id": completion_id, "object": "chat.completion.chunk", "created": created, "model": model, "choices": [{"index": 0, "delta": {"content": chunk}, "finish_reason": None}], } ) events.append( { "id": completion_id, "object": "chat.completion.chunk", "created": created, "model": model, "choices": [{"index": 0, "delta": {}, "finish_reason": None}], "saccadic": payload, } ) events.append( { "id": completion_id, "object": "chat.completion.chunk", "created": created, "model": model, "choices": [ { "index": 0, "delta": {}, "finish_reason": "stop" if payload.get("success") else "error", } ], } ) body = "".join( f"data: {json.dumps(event, ensure_ascii=True, sort_keys=True, separators=(',', ':'))}\n\n" for event in events ) return body + "data: [DONE]\n\n" def _run_artifact(args: argparse.Namespace) -> int: if getattr(args, "compose_speech", False) or getattr(args, "prefer_composed", False): raise ValueError("standalone Python Spark does not support composed speech flags yet") if getattr(args, "recover_on_failure", False): raise ValueError("standalone Python Spark does not support autonomous recovery flags yet") artifact = _load_artifact(args.artifact) avoid_texts = set(str(item) for item in args.avoid_text) if args.mode == "response": payload = _run_response( artifact, str(args.artifact), str(args.text), list(args.host_tool), int(args.max_speech_steps), avoid_texts, bool(args.canonicalize_stream), ) elif args.mode == "history": payload = _run_history( artifact, str(args.artifact), str(args.text), list(args.host_tool), int(args.max_speech_steps), avoid_texts, bool(args.canonicalize_stream), ) elif args.mode == "autonomous": if int(args.turns) > 1: payload = _run_autonomous_turns( str(args.artifact), str(args.text), list(args.host_tool), int(args.max_speech_steps), avoid_texts, bool(args.canonicalize_stream), int(args.turns), args.save_artifact, args.diverse_speech_limit, args.punctuation_floor, ) else: payload = _run_autonomous( artifact, str(args.artifact), str(args.text), list(args.host_tool), int(args.max_speech_steps), avoid_texts, bool(args.canonicalize_stream), ) elif args.mode == "acoustic": payload = _run_acoustic( artifact, str(args.artifact), _parse_audio_samples(args.audio_samples), list(args.host_tool), int(args.max_speech_steps), avoid_texts, float(args.audio_symbol_score_floor), ) else: raise ValueError("standalone Python Spark currently supports response, history, autonomous, and acoustic modes") print(json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":"))) return 0 if payload.get("success") else 1 def _serve_artifact(args: argparse.Namespace) -> int: artifact = _load_artifact(args.artifact) if args.smoke_once: stream = _stream_evidence("Saccadic service runtime smoke.", artifact.get("stream", {})) payload = { "success": True, "service": "sle-artifact-http", "artifact": str(args.artifact), "endpoint": "POST /run", "stream_dynamics": stream, } print(json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":"))) return 0 artifact_path = str(args.artifact) class Handler(BaseHTTPRequestHandler): def log_message(self, _format: str, *values: Any) -> None: return def do_POST(self) -> None: length = int(self.headers.get("Content-Length", "0") or 0) request = json.loads(self.rfile.read(length).decode("utf-8")) if length else {} try: content_type = "application/json" if self.path == "/run": mode = str(request.get("mode", "response")) text = str(request.get("text", "")) host_tools = [str(item) for item in request.get("host_tools", []) or []] max_speech_steps = int(request.get("max_speech_steps", args.max_speech_steps)) avoid_texts = set(str(item) for item in request.get("avoid_texts", []) or []) canonicalize_stream = bool(request.get("canonicalize_stream", False)) if mode == "response": payload = _run_response( artifact, artifact_path, text, host_tools, max_speech_steps, avoid_texts, canonicalize_stream, ) elif mode == "history": payload = _run_history( artifact, artifact_path, text, host_tools, max_speech_steps, avoid_texts, canonicalize_stream, ) elif mode == "autonomous": payload = _run_autonomous( artifact, artifact_path, text, host_tools, max_speech_steps, avoid_texts, canonicalize_stream, ) elif mode == "acoustic": payload = _run_acoustic( artifact, artifact_path, _parse_audio_samples(request.get("audio_samples")), host_tools, max_speech_steps, avoid_texts, float(request.get("audio_symbol_score_floor", 0.25)), ) else: raise ValueError("standalone Python Spark service currently supports response, history, autonomous, and acoustic modes") payload["service"] = "sle-artifact-http" body = json.dumps(payload, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8") self.send_response(200 if payload.get("success") else 422) elif self.path == "/v1/chat/completions": payload = _run_openai_chat_completion( artifact, artifact_path, request, int(args.max_speech_steps), ) model = str(request.get("model", "saccadic")) if bool(request.get("stream", False)): body = _openai_stream_response(model, payload).encode("utf-8") content_type = "text/event-stream; charset=utf-8" else: response = _openai_completion_response(model, payload) body = json.dumps(response, ensure_ascii=True, sort_keys=True, separators=(",", ":")).encode("utf-8") self.send_response(200 if payload.get("success") else 422) else: body = json.dumps({"success": False, "error": "unknown endpoint", "service": "sle-artifact-http"}).encode("utf-8") self.send_response(404) except Exception as exc: body = json.dumps({"success": False, "error": str(exc), "service": "sle-artifact-http"}).encode("utf-8") content_type = "application/json" self.send_response(422) self.send_header("Content-Type", content_type) self.send_header("Content-Length", str(len(body))) self.end_headers() self.wfile.write(body) server = HTTPServer((args.host, int(args.port)), Handler) handled = 0 while args.max_requests is None or handled < int(args.max_requests): server.handle_request() handled += 1 return 0 def _parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(prog="sle_spark.py") subparsers = parser.add_subparsers(dest="command", required=True) run = subparsers.add_parser("run-artifact") run.add_argument("--artifact", default="model.cra") run.add_argument("--mode", choices=("response", "history", "autonomous", "acoustic"), default="response") run.add_argument("--text") run.add_argument("--host-tool", action="append", default=[]) run.add_argument("--max-speech-steps", type=int, default=18) run.add_argument("--avoid-text", action="append", default=[]) run.add_argument("--avoid-replay", action="store_true") run.add_argument("--canonicalize-stream", action="store_true") run.add_argument("--audio-samples") run.add_argument("--audio-symbol-score-floor", type=float, default=0.25) run.add_argument("--turns", type=int, default=1) run.add_argument("--save-artifact", dest="save_artifact") run.add_argument("--output-artifact", dest="save_artifact") run.add_argument("--diverse-speech-limit", type=int) run.add_argument("--punctuation-floor", type=int) run.add_argument("--dt", type=float, default=0.05) run.add_argument("--compose-speech", action="store_true") run.add_argument("--prefer-composed", action="store_true") run.add_argument("--recover-on-failure", action="store_true") run.set_defaults(func=_run_artifact) serve = subparsers.add_parser("serve-artifact") serve.add_argument("--artifact", default="model.cra") serve.add_argument("--host", default="127.0.0.1") serve.add_argument("--port", type=int, default=8765) serve.add_argument("--smoke-once", action="store_true") serve.add_argument("--max-requests", type=int) serve.add_argument("--max-speech-steps", type=int, default=18) serve.add_argument("--dt", type=float, default=0.05) serve.set_defaults(func=_serve_artifact) return parser def main(argv: list[str] | None = None) -> int: args = _parser().parse_args(argv) try: return int(args.func(args)) except Exception as exc: print(json.dumps({"success": False, "error": str(exc)}, ensure_ascii=True, sort_keys=True, separators=(",", ":"))) return 1 if __name__ == "__main__": raise SystemExit(main())