{ "base_model": "Qwen/Qwen3.5-4B", "base_snapshot_revision": "not recorded in retained training configuration", "retained_floating_source_sha256": { "model-00001-of-00002.safetensors": "5da9935a53daa73bf22809a0f89165e8d4406328ecfd6b8c3ac3f38c137c1a56", "model-00002-of-00002.safetensors": "01af619b9a984635330554b09cb45b652673230d55aa568a76d09449f7d066e8" }, "training_method": "LoRA on quantized 4-bit base followed by retained floating-point fusion", "native_unquantized_base_precision": false, "dequantized_from_public_mlx_6bit": false, "training_dataset_public": false, "known_limitation": "Arabic context case adds trailing newline and fails production format check", "nvidia_cuda_inference_verified": false, "amd_rocm_inference_verified": false, "mobile_inference_verified": false, "tested_hardware": "Apple M5 Pro MacBook Pro, 48 GB", "format": "Hugging Face Safetensors", "tensor_roundtrip_exact": 426, "source_dtype_counts": { "mlx.core.bfloat16": 402, "mlx.core.float32": 24 }, "transformations": [ "Remove MLX wrapper prefix", "Transpose depthwise convolution", "Undo offset RMSNorm with exact roundtrip", "Set absent MTP auxiliary layers to zero; no tensors added or removed" ], "runtime_validation": "Native MLX targeted inference; Transformers load without missing or unexpected weights; no full-suite score" }