"""Gradio demo for SupersonicLabs/Julia-1, a 144M typed decision model. Give Julia a game state and any set of named questions -- a choice between 2-20 candidate options, a rubric score, or a yes/no -- and it answers each one with a selected option plus full per-option probabilities. """ import spaces # ZeroGPU: must precede every torch / CUDA-touching import # --------------------------------------------------------------------------- # ZeroGPU main-process shims. # # Julia's loader (julia/cuda.py:configure) runs at module scope on device='cuda' # and calls torch.cuda.set_device() / torch.cuda.reset_peak_memory_stats(). # The ZeroGPU hijack does not patch those two: set_device reaches the # un-emulated torch._C._cuda_setDevice, and reset_peak_memory_stats triggers # _lazy_init -> torch._C._cuda_init, which the hijack replaces with a raise. # Neutralise both until we are inside a real @spaces.GPU worker, where the # hijack is unpatched and the real calls work. # --------------------------------------------------------------------------- import torch _real_cuda_set_device = torch.cuda.set_device _real_reset_peak = torch.cuda.reset_peak_memory_stats torch.cuda.set_device = lambda *a, **k: None torch.cuda.reset_peak_memory_stats = lambda *a, **k: None def _restore_real_cuda(): """Give the GPU worker back the genuine torch.cuda entry points.""" torch.cuda.set_device = _real_cuda_set_device torch.cuda.reset_peak_memory_stats = _real_reset_peak import json import time import gradio as gr from huggingface_hub import snapshot_download from julia import load_model MODEL_ID = "SupersonicLabs/Julia-1" # Julia-1 ships: model.safetensors (577 MB FP32), julia_config.json, # encoder/config.json, tokenizer/{tokenizer.json,tokenizer_config.json}. snapshot = snapshot_download( MODEL_ID, allow_patterns=[ "model.safetensors", "julia_config.json", "encoder/config.json", "tokenizer/*", ], ) # Module-scope eager load. Under the ZeroGPU hijack the .to("cuda") inside the # loader is intercepted: FP32 weights are packed to disk and streamed into # VRAM on the first @spaces.GPU entry. strict_encoding=True enforces the # lossless 48-token-per-option / 8192-token contract from the model card. engine = load_model( snapshot, device="cuda", strict_encoding=True, max_length=8192, head_length=512, ) # The engine is fully initialised now, so the shims have done their job. _restore_real_cuda() # --------------------------------------------------------------------------- # Demo helpers # --------------------------------------------------------------------------- DEFAULT_QUESTIONS = { "team": { "type": "choice", "instructions": "Which team should handle this request?", "criteria": { "billing": "Billing and payment disputes", "shipping": "Shipping and delivery", "access": "Account access and login", }, } } SCHEMA_HINT = ( 'Each question needs "type" ("choice" | "score" | "noul"), ' '"instructions", and "criteria".\n' '- choice: criteria maps option IDs to descriptions (2-20 options)\n' '- score: criteria is an ordered rubric, e.g. ["poor", "ok", "great"]\n' '- noul: no criteria; a true/false question' ) CSS = """ #col-container { max-width: 1080px; margin: 0 auto; } .dark .gradio-container { color: var(--body-text-color); } .answer-head { display: flex; align-items: baseline; gap: 10px; margin: 14px 2px 6px; } .answer-kind { font-size: 12px; font-weight: 600; letter-spacing: .4px; text-transform: uppercase; color: var(--color-accent); opacity: .85; } .answer-pick { font-size: 17px; font-weight: 650; } .bar-row { display: flex; align-items: center; gap: 8px; margin: 3px 0; } .bar-label { flex: 0 0 220px; font-size: 13px; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .bar-track { flex: 1 1 auto; height: 14px; border-radius: 7px; overflow: hidden; background: color-mix(in srgb, var(--body-text-color) 12%, transparent); } .bar-fill { height: 100%; border-radius: 7px; background: var(--color-accent); } .bar-value { flex: 0 0 52px; text-align: right; font-size: 12.5px; font-variant-numeric: tabular-nums; opacity: .85; } .pick .bar-fill { background: linear-gradient(90deg, var(--color-accent), var(--color-accent-soft)); } .score-line { font-size: 14px; margin: 6px 2px; } .score-line b { font-size: 16px; } """ TYPE_LABEL = {"choice": "Choice", "score": "Rubric score", "noul": "Yes / No"} def _bars(labels, probabilities): """Render one question's per-option probabilities as labeled bars.""" best = max(range(len(probabilities)), key=probabilities.__getitem__) rows = [] for i, (label, p) in enumerate(zip(labels, probabilities)): pct = max(0.0, min(1.0, p)) * 100 picked = " pick" if i == best else "" rows.append( f'
' ) return "".join(rows) def _esc(text): return ( str(text) .replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """) ) def _render(result, questions): """Turn engine.predict() output into HTML for the answers panel. ``questions`` is the user's own request, used to recover human-readable labels: choice option IDs get their descriptions, score indices get their rubric entries. """ answers = result.get("answers", {}) if not answers: return 'No answers returned.
' blocks = [] for qid, ans in answers.items(): kind = ans.get("type", "choice") probs = ans.get("probabilities", {}) criteria = (questions.get(qid) or {}).get("criteria") or {} if kind == "choice": label = lambda k: f"{criteria.get(k, k)}" # noqa: E731 elif kind == "score": label = lambda k: criteria[int(k)] if int(k) < len(criteria) else k # noqa: E731 else: label = lambda k: k # noqa: E731 head = ( f'' f"inference: {elapsed:.2f}s · Julia-1 (144M, FP32)
" ) return html, result def _example(state, questions): return [state, json.dumps(questions, ensure_ascii=False, indent=2)] EXAMPLES = [ # The model card's own showcase example. _example( "I was charged twice for the same order.", DEFAULT_QUESTIONS, ), # Router README's intent-routing showcase (translated). _example( "I need to change my password, but the reset email never arrives.", { "intent": { "type": "choice", "instructions": "Which support intent does this message express?", "criteria": { "account": "Account, login and password problems", "orders": "Order status, returns and refunds", "technical": "Website or app malfunctions", "other": "Anything else", }, }, "needs_urgent": { "type": "noul", "instructions": "Is this message urgent and blocking for the user?", }, }, ), # All three question types in one batch (questions are scored independently). _example( "Customer writes: 'The delivery arrived two days late and the box was " "crushed. I want a refund to my original payment method, not store credit.'", { "queue": { "type": "choice", "instructions": "Which queue should handle this complaint?", "criteria": { "logistics": "Damaged or late deliveries", "payments": "Refunds and payment methods", "sales": "Store credit and promotions", }, }, "severity": { "type": "score", "instructions": "Rate the severity of this complaint.", "criteria": ["minor", "moderate", "serious", "critical"], }, "refund_only": { "type": "noul", "instructions": "Is the customer asking only for a refund?", }, }, ), # Rubric scoring only. _example( "Team update: 'We shipped the login redesign. Session drop-off fell " "from 12% to 7% and support tickets about login halved. Two minor " "accessibility bugs remain open.'", { "quality": { "type": "score", "instructions": "Score the quality of this work update.", "criteria": ["poor", "below average", "solid", "excellent"], }, }, ), ] with gr.Blocks(title="Julia-1 · typed decision model") as demo: with gr.Column(elem_id="col-container"): gr.Markdown( f""" # Julia-1 · typed decision model **SupersonicLabs/Julia-1** is a 144M-parameter encoder that turns a *state* plus a set of named questions into decisions: it picks one of 2–20 options, scores a rubric, or answers yes/no — with full probabilities for every option. Ask several questions at once; they are scored independently in one pass. {SCHEMA_HINT.replace(chr(10), " ")} """ ) with gr.Row(): with gr.Column(scale=1): state = gr.Textbox( label="State / context", value="I was charged twice for the same order.", placeholder="Describe the situation, conversation, or game state…", lines=5, ) questions = gr.Code( label="Questions (JSON)", language="json", value=json.dumps(DEFAULT_QUESTIONS, ensure_ascii=False, indent=2), lines=14, ) run = gr.Button("Decide", variant="primary") with gr.Column(scale=1): answers = gr.HTML( label="Answers", value="Answers appear here.
", ) raw = gr.JSON(label="Raw output") with gr.Accordion("About the encoding", open=False): gr.Markdown( "Julia-1 packs `