{ "schema": "karti.model-recipe/v2", "model_name": "Karti-Small-VL-4B", "status": "v1-released-2026-08-31-merged-bf16", "weights_published": true, "private_training_rows_published": false, "base_model": { "repository": "Qwen/Qwen3.5-4B", "revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", "license": "Apache-2.0", "gated": false, "parameters": "4.66B", "model_weights_parameter_count": 4659865088, "architecture": "Qwen3_5ForConditionalGeneration", "selection_gate": "the base's own chat template must render tools, structured tool_calls and the tool role, so that training and serving share one apply_chat_template call", "selection_gate_result": "passed; verified by rendering an image + tool_calls + tool-response conversation through the pinned template", "selection_decision": "among bases passing the gate, chosen for architecture parity with the resident local 27B vision model (same Qwen3_5ForConditionalGeneration serving path) and 4 KV heads versus 8, halving KV cache at a given context", "vision_tower": "depth 24, hidden 1024, out_hidden 2560, patch 16 - configuration identical to Qwen/Qwen3-VL-4B-Instruct", "also_passed_gate_not_selected": { "repository": "Qwen/Qwen3-VL-4B-Instruct", "revision": "ebb281ec70b05090aa6165b016eac8ec08e71b17", "architecture": "Qwen3VLForConditionalGeneration", "reason_not_selected": "a second architecture to carry in parallel - its own serving path and quantization work - for no vision-tower difference" }, "rejected_at_gate": [ "Qwen/Qwen2.5-VL-3B-Instruct", "Qwen/Qwen2.5-VL-7B-Instruct", "microsoft/Phi-4-multimodal-instruct", "OpenGVLab/InternVL3-8B", "HuggingFaceTB/SmolVLM2-2.2B-Instruct", "zai-org/GLM-4.1V-9B-Thinking", "moonshotai/Kimi-VL-A3B-Instruct" ] }, "identity": { "run_id_template": "{family}-{YYYY.MM.DD}-r{NNN}-{manifest_digest:.7}", "release_name_template": "{family}-v{N}", "run_counter": "globally monotonic, never resets", "run_date": "recorded start date; never derived from a week number", "digest_covers": [ "corpus manifest", "base revision", "training hyperparameters" ], "versions_allocated_by": "promotion gate only", "version_is_immutable": true }, "training": { "framework": "TRL", "first_stage": "BF16 LoRA SFT", "vision_tower": "frozen; LoRA applied to the language model", "later_stage": "optional verifier-driven GRPO after held-out improvement", "runs_executed": 0 }, "tool_surface": [ "browser", "cron", "phone", "reply", "route", "tera" ], "tool_surface_note": "every trajectory terminates by delivering through `reply`; no tool executes a confirmed action", "evaluation": { "harness": "a rented GPU Verifiers, containerised agent calling tools over MCP", "train_serve_token_parity": "by construction - one apply_chat_template against the pinned revision", "published_scores": "v0 curated vision_probe only; no training run has been executed", "promotion_policy": "held-out improvement plus owner review", "quality_lane": "canonical BF16 weights on one pinned build and identical hardware", "deployment_lane": "production quantization measured separately after canonical qualification; never mixed into a quality score", "vision_coverage": "a vision suite now exists in kbench and v0 has been measured on it. `vision_probe` is a curated slice mined from public benchmarks (MMStar, RealWorldQA, V*Bench), selected model-independently by category label plus a seeded balanced sample - never by observed difficulty - and scored into its own lane, never averaged into any headline score. A private camera task also exists; it runs only against local services and its score stays private.", "vision_baseline_v0": { "harness": "kbench, run 57ea601a on an internal Blackwell host 2026-08-29", "target": "karti-vl-4b-base@an internal Blackwell host (BF16, thinking off, temperature 0, seed 20260829)", "vision_probe": { "tier": "curated", "n": 220, "accuracy": 0.727273, "stderr": 0.030095, "dataset_fingerprint": "sha256:5cc4ba628a20a2b2" }, "contamination_canary": { "tier": "canary", "n": 2, "accuracy": 0.0, "reading": "clean" }, "private_signal_task": { "name": "camera_watch", "n": 120, "score_published": false, "why": "scored on frames from the home's own cameras; the task refuses any target that is not a local service and the frames never leave our hardware, so the aggregate stays private" }, "note": "this is the UNTRAINED base. It is the number v1 has to beat, and it is not a claim that the model is good at vision - only that we can now tell whether training moved it." } }, "plan": [ "control run: existing text corpus on this base, reported with the serialization caveat", "vision rows: images paired with tool-calling targets, frozen vision tower", "topology: decide on evidence whether this subsumes the text-only 3B" ], "carried_differences": { "tool_call_serialization": { "format": "value inside tags; not a JSON object", "parity_holds": true, "parity_basis": "structured tool_calls are passed to apply_chat_template and never hand-formatted in either shape", "consequence": "the control run cannot claim the 3B text corpus reaches the model unchanged; it measures base plus serialization together and must be reported that way" }, "thinking_default_on": { "behaviour": "the template opens a block and the generation prompt ends inside one", "observed_failure_mode": "under a small max_tokens budget the reasoning block consumes the whole allowance and content returns empty with finish_reason=length", "mitigation": "enable_thinking:false pinned for every run and serving config", "required_canary": "a tight-max_tokens request must return non-empty content before any result from this model is believed" } }, "releases": { "v0": { "role": "reference baseline; the pinned base measured on our hardware before any training", "weights_repository": "Qwen/Qwen3.5-4B", "weights_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", "weights_mirrored_here": true, "weights_mirror_note": "mirrored in this repository and byte-identical to the pinned upstream revision; both shard SHA-256 digests match Qwen/Qwen3.5-4B @ 851bf6e8. Tagged v0; the exact pre-v1 repository state is tagged pre-v1.", "precision": "BF16", "training_applied": "none", "gate_passed": "n/a - a baseline is not gated; v1 onward are", "measured_on": "DGX Spark GB10, 121.7 GB unified, vLLM 0.27.1, max_model_len 32768, kv-cache fp8, temperature 0, single stream", "measured_2026_08_29": { "tok_per_s_128": 18.24, "tok_per_s_512": 17.74, "vision": "correct on synthetic shape/colour image", "tool_call": "parsed via qwen3_xml", "canary_19x23": 437, "canary_empty_content_at_max_tokens_80": "non-empty, finish_reason=stop" }, "observation": "at BF16 this 4B runs at roughly the same single-stream rate as the 27B NVFP4 on the same box - a memory-bandwidth result, and the argument for quantizing the deployment lane" } }, "serving": { "aliases": [ "karti-vl", "karti-vl-4b-base" ], "port": 8002, "runtime": "vLLM 0.27.1", "tool_call_parser": "qwen3_xml", "chat_template": "the model's own; never overridden - train/serve parity depends on it", "enable_thinking": false }, "current_version": { "release_name": "Karti-Small-VL-4B-v1", "version": 1, "kind": "merged BF16 (LoRA baked in; the adapter is deliberately not published)", "training": "LoRA r16/alpha32, lr 5e-5, 75 steps, 248 language modules, vision tower frozen (297 tensors byte-identical to base)", "corpus_rows": 3569, "corpus_abstention_share": 0.144, "gates": { "invented_identifier": { "v0": 0.378, "v1": 0.0232 }, "lane_accuracy": { "v0": 0.595, "v1": 0.9668 }, "vizwiz_false_refusal": { "v0": 0.22, "v1": 0.227 }, "screenspot_overall": { "v0": 0.718, "v1": 0.907 }, "screenspot_small": { "v0": 0.58, "v1": 0.883 }, "vision_probe": { "v0": 0.727, "v1": 0.7318 } }, "caveats": [ "much of the ScreenSpot gain is learning the normalised 0-1000 convention shared by GUI-Odyssey (trained) and ScreenSpot (held out); unparseable replies fell 48 -> 9", "VizWiz recall moved only ~1.2 sigma: v1 does not read photographs better than v0, it merely does not over-refuse", "abstain_illegible still fabricates 7 of 112" ], "serving_warning": "serve the merged weights, not a LoRA adapter: vLLM 0.27.1 applies this adapter incompletely, costing about 0.19 absolute ScreenSpot accuracy", "measured_on": "DGX Spark GB10, vLLM 0.27.1, max_model_len 32768, kv-cache fp8, temperature 0, single stream, best of 3", "measured_2026_08_31": { "tok_per_s_128": 21.01, "tok_per_s_512": 21.08, "note": "v1 BF16. The NVFP4 build measures 51.7/51.8 on the same box and method, a 2.5x speedup." } }, "published_scores": "v1 gates measured against v0 on the same endpoint and decode path; see the model card" }