{
"schema": "karti.model-recipe/v2",
"model_name": "Karti-Small-VL-4B",
"status": "v1-released-2026-08-31-merged-bf16",
"weights_published": true,
"private_training_rows_published": false,
"base_model": {
"repository": "Qwen/Qwen3.5-4B",
"revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a",
"license": "Apache-2.0",
"gated": false,
"parameters": "4.66B",
"model_weights_parameter_count": 4659865088,
"architecture": "Qwen3_5ForConditionalGeneration",
"selection_gate": "the base's own chat template must render tools, structured tool_calls and the tool role, so that training and serving share one apply_chat_template call",
"selection_gate_result": "passed; verified by rendering an image + tool_calls + tool-response conversation through the pinned template",
"selection_decision": "among bases passing the gate, chosen for architecture parity with the resident local 27B vision model (same Qwen3_5ForConditionalGeneration serving path) and 4 KV heads versus 8, halving KV cache at a given context",
"vision_tower": "depth 24, hidden 1024, out_hidden 2560, patch 16 - configuration identical to Qwen/Qwen3-VL-4B-Instruct",
"also_passed_gate_not_selected": {
"repository": "Qwen/Qwen3-VL-4B-Instruct",
"revision": "ebb281ec70b05090aa6165b016eac8ec08e71b17",
"architecture": "Qwen3VLForConditionalGeneration",
"reason_not_selected": "a second architecture to carry in parallel - its own serving path and quantization work - for no vision-tower difference"
},
"rejected_at_gate": [
"Qwen/Qwen2.5-VL-3B-Instruct",
"Qwen/Qwen2.5-VL-7B-Instruct",
"microsoft/Phi-4-multimodal-instruct",
"OpenGVLab/InternVL3-8B",
"HuggingFaceTB/SmolVLM2-2.2B-Instruct",
"zai-org/GLM-4.1V-9B-Thinking",
"moonshotai/Kimi-VL-A3B-Instruct"
]
},
"identity": {
"run_id_template": "{family}-{YYYY.MM.DD}-r{NNN}-{manifest_digest:.7}",
"release_name_template": "{family}-v{N}",
"run_counter": "globally monotonic, never resets",
"run_date": "recorded start date; never derived from a week number",
"digest_covers": [
"corpus manifest",
"base revision",
"training hyperparameters"
],
"versions_allocated_by": "promotion gate only",
"version_is_immutable": true
},
"training": {
"framework": "TRL",
"first_stage": "BF16 LoRA SFT",
"vision_tower": "frozen; LoRA applied to the language model",
"later_stage": "optional verifier-driven GRPO after held-out improvement",
"runs_executed": 0
},
"tool_surface": [
"browser",
"cron",
"phone",
"reply",
"route",
"tera"
],
"tool_surface_note": "every trajectory terminates by delivering through `reply`; no tool executes a confirmed action",
"evaluation": {
"harness": "a rented GPU Verifiers, containerised agent calling tools over MCP",
"train_serve_token_parity": "by construction - one apply_chat_template against the pinned revision",
"published_scores": "v0 curated vision_probe only; no training run has been executed",
"promotion_policy": "held-out improvement plus owner review",
"quality_lane": "canonical BF16 weights on one pinned build and identical hardware",
"deployment_lane": "production quantization measured separately after canonical qualification; never mixed into a quality score",
"vision_coverage": "a vision suite now exists in kbench and v0 has been measured on it. `vision_probe` is a curated slice mined from public benchmarks (MMStar, RealWorldQA, V*Bench), selected model-independently by category label plus a seeded balanced sample - never by observed difficulty - and scored into its own lane, never averaged into any headline score. A private camera task also exists; it runs only against local services and its score stays private.",
"vision_baseline_v0": {
"harness": "kbench, run 57ea601a on an internal Blackwell host 2026-08-29",
"target": "karti-vl-4b-base@an internal Blackwell host (BF16, thinking off, temperature 0, seed 20260829)",
"vision_probe": {
"tier": "curated",
"n": 220,
"accuracy": 0.727273,
"stderr": 0.030095,
"dataset_fingerprint": "sha256:5cc4ba628a20a2b2"
},
"contamination_canary": {
"tier": "canary",
"n": 2,
"accuracy": 0.0,
"reading": "clean"
},
"private_signal_task": {
"name": "camera_watch",
"n": 120,
"score_published": false,
"why": "scored on frames from the home's own cameras; the task refuses any target that is not a local service and the frames never leave our hardware, so the aggregate stays private"
},
"note": "this is the UNTRAINED base. It is the number v1 has to beat, and it is not a claim that the model is good at vision - only that we can now tell whether training moved it."
}
},
"plan": [
"control run: existing text corpus on this base, reported with the serialization caveat",
"vision rows: images paired with tool-calling targets, frozen vision tower",
"topology: decide on evidence whether this subsumes the text-only 3B"
],
"carried_differences": {
"tool_call_serialization": {
"format": "value inside tags; not a JSON object",
"parity_holds": true,
"parity_basis": "structured tool_calls are passed to apply_chat_template and never hand-formatted in either shape",
"consequence": "the control run cannot claim the 3B text corpus reaches the model unchanged; it measures base plus serialization together and must be reported that way"
},
"thinking_default_on": {
"behaviour": "the template opens a block and the generation prompt ends inside one",
"observed_failure_mode": "under a small max_tokens budget the reasoning block consumes the whole allowance and content returns empty with finish_reason=length",
"mitigation": "enable_thinking:false pinned for every run and serving config",
"required_canary": "a tight-max_tokens request must return non-empty content before any result from this model is believed"
}
},
"releases": {
"v0": {
"role": "reference baseline; the pinned base measured on our hardware before any training",
"weights_repository": "Qwen/Qwen3.5-4B",
"weights_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a",
"weights_mirrored_here": true,
"weights_mirror_note": "mirrored in this repository and byte-identical to the pinned upstream revision; both shard SHA-256 digests match Qwen/Qwen3.5-4B @ 851bf6e8. Tagged v0; the exact pre-v1 repository state is tagged pre-v1.",
"precision": "BF16",
"training_applied": "none",
"gate_passed": "n/a - a baseline is not gated; v1 onward are",
"measured_on": "DGX Spark GB10, 121.7 GB unified, vLLM 0.27.1, max_model_len 32768, kv-cache fp8, temperature 0, single stream",
"measured_2026_08_29": {
"tok_per_s_128": 18.24,
"tok_per_s_512": 17.74,
"vision": "correct on synthetic shape/colour image",
"tool_call": "parsed via qwen3_xml",
"canary_19x23": 437,
"canary_empty_content_at_max_tokens_80": "non-empty, finish_reason=stop"
},
"observation": "at BF16 this 4B runs at roughly the same single-stream rate as the 27B NVFP4 on the same box - a memory-bandwidth result, and the argument for quantizing the deployment lane"
}
},
"serving": {
"aliases": [
"karti-vl",
"karti-vl-4b-base"
],
"port": 8002,
"runtime": "vLLM 0.27.1",
"tool_call_parser": "qwen3_xml",
"chat_template": "the model's own; never overridden - train/serve parity depends on it",
"enable_thinking": false
},
"current_version": {
"release_name": "Karti-Small-VL-4B-v1",
"version": 1,
"kind": "merged BF16 (LoRA baked in; the adapter is deliberately not published)",
"training": "LoRA r16/alpha32, lr 5e-5, 75 steps, 248 language modules, vision tower frozen (297 tensors byte-identical to base)",
"corpus_rows": 3569,
"corpus_abstention_share": 0.144,
"gates": {
"invented_identifier": {
"v0": 0.378,
"v1": 0.0232
},
"lane_accuracy": {
"v0": 0.595,
"v1": 0.9668
},
"vizwiz_false_refusal": {
"v0": 0.22,
"v1": 0.227
},
"screenspot_overall": {
"v0": 0.718,
"v1": 0.907
},
"screenspot_small": {
"v0": 0.58,
"v1": 0.883
},
"vision_probe": {
"v0": 0.727,
"v1": 0.7318
}
},
"caveats": [
"much of the ScreenSpot gain is learning the normalised 0-1000 convention shared by GUI-Odyssey (trained) and ScreenSpot (held out); unparseable replies fell 48 -> 9",
"VizWiz recall moved only ~1.2 sigma: v1 does not read photographs better than v0, it merely does not over-refuse",
"abstain_illegible still fabricates 7 of 112"
],
"serving_warning": "serve the merged weights, not a LoRA adapter: vLLM 0.27.1 applies this adapter incompletely, costing about 0.19 absolute ScreenSpot accuracy",
"measured_on": "DGX Spark GB10, vLLM 0.27.1, max_model_len 32768, kv-cache fp8, temperature 0, single stream, best of 3",
"measured_2026_08_31": {
"tok_per_s_128": 21.01,
"tok_per_s_512": 21.08,
"note": "v1 BF16. The NVFP4 build measures 51.7/51.8 on the same box and method, a 2.5x speedup."
}
},
"published_scores": "v1 gates measured against v0 on the same endpoint and decode path; see the model card"
}