"""Gradio demo for SupersonicLabs/Julia-1, a 144M typed decision model. Give Julia a game state and any set of named questions -- a choice between 2-20 candidate options, a rubric score, or a yes/no -- and it answers each one with a selected option plus full per-option probabilities. """ import spaces # ZeroGPU: must precede every torch / CUDA-touching import # --------------------------------------------------------------------------- # ZeroGPU main-process shims. # # Julia's loader (julia/cuda.py:configure) runs at module scope on device='cuda' # and calls torch.cuda.set_device() / torch.cuda.reset_peak_memory_stats(). # The ZeroGPU hijack does not patch those two: set_device reaches the # un-emulated torch._C._cuda_setDevice, and reset_peak_memory_stats triggers # _lazy_init -> torch._C._cuda_init, which the hijack replaces with a raise. # Neutralise both until we are inside a real @spaces.GPU worker, where the # hijack is unpatched and the real calls work. # --------------------------------------------------------------------------- import torch _real_cuda_set_device = torch.cuda.set_device _real_reset_peak = torch.cuda.reset_peak_memory_stats torch.cuda.set_device = lambda *a, **k: None torch.cuda.reset_peak_memory_stats = lambda *a, **k: None def _restore_real_cuda(): """Give the GPU worker back the genuine torch.cuda entry points.""" torch.cuda.set_device = _real_cuda_set_device torch.cuda.reset_peak_memory_stats = _real_reset_peak import json import time import gradio as gr from huggingface_hub import snapshot_download from julia import load_model MODEL_ID = "SupersonicLabs/Julia-1" # Julia-1 ships: model.safetensors (577 MB FP32), julia_config.json, # encoder/config.json, tokenizer/{tokenizer.json,tokenizer_config.json}. snapshot = snapshot_download( MODEL_ID, allow_patterns=[ "model.safetensors", "julia_config.json", "encoder/config.json", "tokenizer/*", ], ) # Module-scope eager load. Under the ZeroGPU hijack the .to("cuda") inside the # loader is intercepted: FP32 weights are packed to disk and streamed into # VRAM on the first @spaces.GPU entry. strict_encoding=True enforces the # lossless 48-token-per-option / 8192-token contract from the model card. engine = load_model( snapshot, device="cuda", strict_encoding=True, max_length=8192, head_length=512, ) # The engine is fully initialised now, so the shims have done their job. _restore_real_cuda() # --------------------------------------------------------------------------- # Demo helpers # --------------------------------------------------------------------------- DEFAULT_QUESTIONS = { "team": { "type": "choice", "instructions": "Which team should handle this request?", "criteria": { "billing": "Billing and payment disputes", "shipping": "Shipping and delivery", "access": "Account access and login", }, } } SCHEMA_HINT = ( 'Each question needs "type" ("choice" | "score" | "noul"), ' '"instructions", and "criteria".\n' '- choice: criteria maps option IDs to descriptions (2-20 options)\n' '- score: criteria is an ordered rubric, e.g. ["poor", "ok", "great"]\n' '- noul: no criteria; a true/false question' ) CSS = """ #col-container { max-width: 1080px; margin: 0 auto; } .dark .gradio-container { color: var(--body-text-color); } .answer-head { display: flex; align-items: baseline; gap: 10px; margin: 14px 2px 6px; } .answer-kind { font-size: 12px; font-weight: 600; letter-spacing: .4px; text-transform: uppercase; color: var(--color-accent); opacity: .85; } .answer-pick { font-size: 17px; font-weight: 650; } .bar-row { display: flex; align-items: center; gap: 8px; margin: 3px 0; } .bar-label { flex: 0 0 220px; font-size: 13px; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .bar-track { flex: 1 1 auto; height: 14px; border-radius: 7px; overflow: hidden; background: color-mix(in srgb, var(--body-text-color) 12%, transparent); } .bar-fill { height: 100%; border-radius: 7px; background: var(--color-accent); } .bar-value { flex: 0 0 52px; text-align: right; font-size: 12.5px; font-variant-numeric: tabular-nums; opacity: .85; } .pick .bar-fill { background: linear-gradient(90deg, var(--color-accent), var(--color-accent-soft)); } .score-line { font-size: 14px; margin: 6px 2px; } .score-line b { font-size: 16px; } """ TYPE_LABEL = {"choice": "Choice", "score": "Rubric score", "noul": "Yes / No"} def _bars(labels, probabilities): """Render one question's per-option probabilities as labeled bars.""" best = max(range(len(probabilities)), key=probabilities.__getitem__) rows = [] for i, (label, p) in enumerate(zip(labels, probabilities)): pct = max(0.0, min(1.0, p)) * 100 picked = " pick" if i == best else "" rows.append( f'
' f'
{_esc(label)}
' f'
' f'
{p * 100:.1f}%
' ) return "".join(rows) def _esc(text): return ( str(text) .replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """) ) def _render(result, questions): """Turn engine.predict() output into HTML for the answers panel. ``questions`` is the user's own request, used to recover human-readable labels: choice option IDs get their descriptions, score indices get their rubric entries. """ answers = result.get("answers", {}) if not answers: return '

No answers returned.

' blocks = [] for qid, ans in answers.items(): kind = ans.get("type", "choice") probs = ans.get("probabilities", {}) criteria = (questions.get(qid) or {}).get("criteria") or {} if kind == "choice": label = lambda k: f"{criteria.get(k, k)}" # noqa: E731 elif kind == "score": label = lambda k: criteria[int(k)] if int(k) < len(criteria) else k # noqa: E731 else: label = lambda k: k # noqa: E731 head = ( f'
' f'{TYPE_LABEL.get(kind, kind)}' f'{_esc(qid)}
' ) if kind == "choice": pick = ans.get("choice", "?") body = ( f'
Selected: {_esc(pick)}' + (f" — {_esc(criteria.get(pick, ''))}" if criteria.get(pick) else "") + "
" + _bars([label(k) for k in probs], list(probs.values())) ) elif kind == "score": body = ( f'
Expected score: {ans.get("score", 0):.2f}' f' on a 0-{len(probs) - 1} rubric
' + _bars([label(k) for k in probs], list(probs.values())) ) else: # noul yes = ans.get("noul", 0.0) body = ( f'
P(yes) = {yes * 100:.1f}%' f'  ·  P(no) = {(1 - yes) * 100:.1f}%
' ) blocks.append(head + body) return "".join(blocks) def _parse_questions(raw): """Validate the questions JSON before it reaches the engine.""" questions = json.loads(raw) if not isinstance(questions, dict) or not questions: raise ValueError("Questions must be a non-empty JSON object mapping IDs to questions.") if len(questions) > 50: raise ValueError("At most 50 questions per request (they are scored in one batch).") for qid, q in questions.items(): if not isinstance(qid, str) or not qid or not isinstance(q, dict): raise ValueError(f"Question {qid!r}: needs a non-empty string ID and an object body.") kind = q.get("type") if kind not in ("choice", "score", "noul"): raise ValueError(f"Question {qid!r}: type must be choice, score, or noul.") if not isinstance(q.get("instructions"), str) or not q["instructions"].strip(): raise ValueError(f"Question {qid!r}: instructions must be non-empty text.") criteria = q.get("criteria") if kind == "choice": if not isinstance(criteria, dict) or not 2 <= len(criteria) <= 20: raise ValueError(f"Question {qid!r}: choice criteria need 2-20 options.") if any(not isinstance(k, str) or not k or not isinstance(v, str) or not v for k, v in criteria.items()): raise ValueError(f"Question {qid!r}: choice criteria map IDs to descriptions.") elif kind == "score": if not isinstance(criteria, list) or not 2 <= len(criteria) <= 20: raise ValueError(f"Question {qid!r}: score needs an ordered rubric list.") elif "criteria" in q: raise ValueError(f"Question {qid!r}: noul takes no criteria.") return questions @spaces.GPU(duration=10) # measured: 0.34s warm, ~3s cold attach, 0.46s 3-question batch def decide(state: str, questions_json: str): """Answer typed decision questions about a state. Args: state: The game state / context / situation text (up to ~8k tokens). questions_json: JSON object of named questions. Each question has "type" ("choice", "score", or "noul"), "instructions", and "criteria" (option map for choice, ordered rubric for score, absent for noul). Returns: Rendered HTML answers and the raw JSON result. """ questions = _parse_questions(questions_json) start = time.perf_counter() result = engine.predict(state=state, questions=questions) elapsed = time.perf_counter() - start html = _render(result, questions) + ( f'

' f"inference: {elapsed:.2f}s · Julia-1 (144M, FP32)

" ) return html, result def _example(state, questions): return [state, json.dumps(questions, ensure_ascii=False, indent=2)] EXAMPLES = [ # The model card's own showcase example. _example( "I was charged twice for the same order.", DEFAULT_QUESTIONS, ), # Router README's intent-routing showcase (translated). _example( "I need to change my password, but the reset email never arrives.", { "intent": { "type": "choice", "instructions": "Which support intent does this message express?", "criteria": { "account": "Account, login and password problems", "orders": "Order status, returns and refunds", "technical": "Website or app malfunctions", "other": "Anything else", }, }, "needs_urgent": { "type": "noul", "instructions": "Is this message urgent and blocking for the user?", }, }, ), # All three question types in one batch (questions are scored independently). _example( "Customer writes: 'The delivery arrived two days late and the box was " "crushed. I want a refund to my original payment method, not store credit.'", { "queue": { "type": "choice", "instructions": "Which queue should handle this complaint?", "criteria": { "logistics": "Damaged or late deliveries", "payments": "Refunds and payment methods", "sales": "Store credit and promotions", }, }, "severity": { "type": "score", "instructions": "Rate the severity of this complaint.", "criteria": ["minor", "moderate", "serious", "critical"], }, "refund_only": { "type": "noul", "instructions": "Is the customer asking only for a refund?", }, }, ), # Rubric scoring only. _example( "Team update: 'We shipped the login redesign. Session drop-off fell " "from 12% to 7% and support tickets about login halved. Two minor " "accessibility bugs remain open.'", { "quality": { "type": "score", "instructions": "Score the quality of this work update.", "criteria": ["poor", "below average", "solid", "excellent"], }, }, ), ] with gr.Blocks(title="Julia-1 · typed decision model") as demo: with gr.Column(elem_id="col-container"): gr.Markdown( f""" # Julia-1 · typed decision model **SupersonicLabs/Julia-1** is a 144M-parameter encoder that turns a *state* plus a set of named questions into decisions: it picks one of 2–20 options, scores a rubric, or answers yes/no — with full probabilities for every option. Ask several questions at once; they are scored independently in one pass. {SCHEMA_HINT.replace(chr(10), " ")} """ ) with gr.Row(): with gr.Column(scale=1): state = gr.Textbox( label="State / context", value="I was charged twice for the same order.", placeholder="Describe the situation, conversation, or game state…", lines=5, ) questions = gr.Code( label="Questions (JSON)", language="json", value=json.dumps(DEFAULT_QUESTIONS, ensure_ascii=False, indent=2), lines=14, ) run = gr.Button("Decide", variant="primary") with gr.Column(scale=1): answers = gr.HTML( label="Answers", value="

Answers appear here.

", ) raw = gr.JSON(label="Raw output") with gr.Accordion("About the encoding", open=False): gr.Markdown( "Julia-1 packs `