Download neuron_config.json from jburtoft/Laguna-XS2-neuron-compiled: direct link, hf CLI and curl.
- Browser
- Download file 12.3 kB
-
https://huggingface.co/jburtoft/Laguna-XS2-neuron-compiled/resolve/main/neuron_config.json
- Command line
-
hf download hf://jburtoft/Laguna-XS2-neuron-compiled/neuron_config.json
-
curl -L -o neuron_config.json https://huggingface.co/jburtoft/Laguna-XS2-neuron-compiled/resolve/main/neuron_config.json
12.3 kB
Invalid JSON:Unexpected token 'I', ..."p_bound": Infinity,
"... is not valid JSON
| { | |
| "_name_or_path": "/mnt/models/Laguna-XS.2/", | |
| "add_cross_attention": false, | |
| "architectures": [ | |
| "LagunaForCausalLM" | |
| ], | |
| "attention_bias": false, | |
| "attention_dropout": 0.0, | |
| "attribute_map": {}, | |
| "auto_map": { | |
| "AutoConfig": "configuration_laguna.LagunaConfig", | |
| "AutoModelForCausalLM": "modeling_laguna.LagunaForCausalLM" | |
| }, | |
| "bad_words_ids": null, | |
| "begin_suppress_tokens": null, | |
| "bos_token_id": 2, | |
| "chunk_size_feed_forward": 0, | |
| "cross_attention_hidden_size": null, | |
| "decoder_start_token_id": null, | |
| "diversity_penalty": 0.0, | |
| "do_sample": false, | |
| "early_stopping": false, | |
| "encoder_no_repeat_ngram_size": 0, | |
| "eos_token_id": [ | |
| 2, | |
| 24 | |
| ], | |
| "exponential_decay_length_penalty": null, | |
| "finetuning_task": null, | |
| "forced_bos_token_id": null, | |
| "forced_eos_token_id": null, | |
| "fused_spec_config": null, | |
| "gating": true, | |
| "head_dim": 128, | |
| "hidden_act": "silu", | |
| "hidden_size": 2048, | |
| "id2label": { | |
| "0": "LABEL_0", | |
| "1": "LABEL_1" | |
| }, | |
| "intermediate_size": 8192, | |
| "is_decoder": false, | |
| "is_encoder_decoder": false, | |
| "label2id": { | |
| "LABEL_0": 0, | |
| "LABEL_1": 1 | |
| }, | |
| "layer_types": [ | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "full_attention", | |
| "sliding_attention", | |
| "sliding_attention", | |
| "sliding_attention" | |
| ], | |
| "length_penalty": 1.0, | |
| "max_length": 20, | |
| "max_position_embeddings": 131072, | |
| "metadata": null, | |
| "min_length": 0, | |
| "mlp_layer_types": [ | |
| "dense", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse", | |
| "sparse" | |
| ], | |
| "model_type": "", | |
| "moe_apply_router_weight_on_input": false, | |
| "moe_cte_ep_degree": 1, | |
| "moe_cte_tp_degree": 4, | |
| "moe_intermediate_size": 512, | |
| "moe_routed_scaling_factor": 2.5, | |
| "moe_tkg_ep_degree": 1, | |
| "moe_tkg_tp_degree": 4, | |
| "n_shared_experts": 1, | |
| "neuron_config": { | |
| "activation_quantization_type": null, | |
| "allow_input_truncation": false, | |
| "apply_seq_ids_mask": false, | |
| "async_mode": false, | |
| "attention_dp_degree": 1, | |
| "attention_dtype": null, | |
| "attn_block_cte_nki_kernel_enabled": false, | |
| "attn_block_tkg_nki_kernel_cache_update": false, | |
| "attn_block_tkg_nki_kernel_cascaded_attention": false, | |
| "attn_block_tkg_nki_kernel_disable_gpsimd_sb2sb": false, | |
| "attn_block_tkg_nki_kernel_enabled": false, | |
| "attn_block_tkg_nki_kernel_use_online_softmax": true, | |
| "attn_cls": "NeuronLlamaAttention", | |
| "attn_kernel_enabled": true, | |
| "attn_tkg_builtin_kernel_enabled": false, | |
| "attn_tkg_nki_kernel_enabled": false, | |
| "batch_size": 4, | |
| "blockwise_matmul_config": { | |
| "always_augment_inputs_for_blockwise_matmul": false, | |
| "block_sharding_strategy": { | |
| "__objclass__": { | |
| "__module__": "nkilib.core.moe.moe_cte.moe_cte_utils", | |
| "__name__": "BlockShardStrategy" | |
| }, | |
| "_name_": "HI_LO", | |
| "_sort_order_": 0, | |
| "_value_": 0 | |
| }, | |
| "block_size": 512, | |
| "blockwise_nki_autograd_cls": null, | |
| "logical_nc_config": { | |
| "__objclass__": { | |
| "__module__": "neuronx_distributed.utils.model_utils", | |
| "__name__": "LogicalNCConfig" | |
| }, | |
| "_name_": "LNC_2", | |
| "_sort_order_": 1, | |
| "_value_": 2 | |
| }, | |
| "num_static_blocks": null, | |
| "optimized_block_to_token_mapping": true, | |
| "pad_num_blocks_to_even": false, | |
| "parallelize_token_to_block_mapping": true, | |
| "skip_dma_token": false, | |
| "skip_dma_weight": false, | |
| "use_block_parallel": false, | |
| "use_shard_on_block_dynamic_while": false, | |
| "use_shard_on_intermediate_dynamic_while": false, | |
| "use_torch_block_wise": true | |
| }, | |
| "bucket_n_active_tokens": false, | |
| "buckets": [ | |
| 4096 | |
| ], | |
| "capacity_factor": null, | |
| "cast_type": "config", | |
| "cc_pipeline_tiling_factor": 2, | |
| "chunked_prefill_config": null, | |
| "context_encoding_buckets": null, | |
| "cp_degree": 1, | |
| "ctx_batch_size": 1, | |
| "disable_argmax_kernel": false, | |
| "disable_kv_cache_tiling": false, | |
| "disable_numeric_cc_token": false, | |
| "dma_order_config": null, | |
| "draft_model_modules_to_not_convert": null, | |
| "eagle_rolling_buffer_kernel_enabled": false, | |
| "early_expert_affinity_modulation": false, | |
| "enable_bucketing": true, | |
| "enable_cte_modular_flow": false, | |
| "enable_eagle_draft_input_norm": false, | |
| "enable_eagle_speculation": false, | |
| "enable_fused_speculation": false, | |
| "enable_long_context_mode": false, | |
| "enable_output_completion_notifications": false, | |
| "enable_spill_reload_dge": false, | |
| "enable_token_tree": false, | |
| "enable_ve_data_parallel": false, | |
| "ep_degree": 1, | |
| "ep_dispatch_cc_option": "AR_AG", | |
| "expert_mlp_nki_kernel_enabled": null, | |
| "flash_decoding_enabled": false, | |
| "fused_qkv": false, | |
| "fused_rmsnorm_skip_gamma": false, | |
| "fused_shared_experts": false, | |
| "gate_clamp_lower_limit": null, | |
| "gate_clamp_upper_limit": null, | |
| "glu_mlp": true, | |
| "glu_type": "glu", | |
| "hidden_act_bias": 0.0, | |
| "hidden_act_scaling_factor": 1.0, | |
| "hybrid_sharding_config": null, | |
| "is_block_kv_layout": false, | |
| "is_chunked_prefill": false, | |
| "is_continuous_batching": true, | |
| "is_eagle3": false, | |
| "is_eagle_draft": false, | |
| "is_full_model_shuffled": false, | |
| "is_hidden_dim_shuffled": false, | |
| "is_intermediate_dim_shuffled": false, | |
| "is_medusa": false, | |
| "is_mxfp4_compute": false, | |
| "is_prefill_stage": null, | |
| "is_prefix_caching": false, | |
| "k_cache_transposed": false, | |
| "kv_cache_batch_size": 4, | |
| "kv_cache_padding_size": 0, | |
| "kv_cache_quant": false, | |
| "kv_cache_tiling": false, | |
| "kv_cache_update_with_kernel": false, | |
| "kv_quant_config": null, | |
| "layer_boundary_markers": false, | |
| "lm_head_pad": true, | |
| "lm_head_pad_alignment_size": 1, | |
| "local_ranks_size": 4, | |
| "logical_nc_config": 2, | |
| "lora_config": null, | |
| "max_batch_size": 4, | |
| "max_context_length": 4096, | |
| "max_length": 4096, | |
| "max_new_tokens": null, | |
| "medusa_speculation_length": 0, | |
| "medusa_tree": null, | |
| "mlp_cp_degree": 1, | |
| "mlp_kernel_enabled": false, | |
| "mlp_kernel_fuse_residual_add": false, | |
| "mlp_tkg_nki_kernel_enabled": false, | |
| "modules_to_not_convert": null, | |
| "moe_ep_degree": 1, | |
| "moe_fused_nki_kernel_enabled": true, | |
| "moe_mask_padded_tokens": false, | |
| "moe_tp_degree": 4, | |
| "n_active_tokens": 4096, | |
| "n_positions": 4096, | |
| "normalize_top_k_affinities": true, | |
| "num_medusa_heads": 0, | |
| "on_cpu": false, | |
| "on_device_sampling_config": { | |
| "deterministic": false, | |
| "do_sample": false, | |
| "dynamic": true, | |
| "global_topk": 256, | |
| "on_device_sampling_config": true, | |
| "sampling_dp_degree": 1, | |
| "temperature": 1.0, | |
| "top_k": 1, | |
| "top_k_kernel_enabled": true, | |
| "top_p": 1.0 | |
| }, | |
| "out_proj_kernel_enabled": false, | |
| "output_logits": false, | |
| "overrides_torch_dtype": true, | |
| "pa_block_size": 4096, | |
| "pa_num_blocks": 2, | |
| "padded_hidden_size": null, | |
| "padded_intermediate_size": null, | |
| "padding_side": "right", | |
| "pp_degree": 1, | |
| "pre_rope_rmsnorm": false, | |
| "prefix_buckets": null, | |
| "qk_layernorm": false, | |
| "qkv_cte_nki_kernel_fuse_rope": false, | |
| "qkv_kernel_enabled": false, | |
| "qkv_kernel_fuse_residual_add": false, | |
| "qkv_kernel_nbsd_layout": false, | |
| "qkv_nki_kernel_enabled": false, | |
| "quantization_block_axis": null, | |
| "quantization_block_size": null, | |
| "quantization_dtype": "int8", | |
| "quantization_scale_dtype": "f32", | |
| "quantization_type": "per_tensor_symmetric", | |
| "quantize_clamp_bound": Infinity, | |
| "quantized": false, | |
| "quantized_checkpoints_path": null, | |
| "quantized_mlp_kernel_enabled": false, | |
| "return_expert_index": false, | |
| "return_router_logits": false, | |
| "rmsnorm_quantize_kernel_enabled": false, | |
| "router_config": { | |
| "act_fn": "sigmoid", | |
| "dtype": "float32" | |
| }, | |
| "router_topk_nki_kernel_enabled": null, | |
| "rpl_reduce_dtype": null, | |
| "save_sharded_checkpoint": true, | |
| "scratchpad_page_size": null, | |
| "seq_len": 4096, | |
| "seq_len_threshold_for_cc_tiling": 16384, | |
| "sequence_parallel_enabled": false, | |
| "shared_experts_sequence_parallel_enabled": false, | |
| "shared_mlp_nki_kernel_enabled": null, | |
| "skip_sharding": false, | |
| "skip_warmup": false, | |
| "spec_batch_size": 4, | |
| "speculation_length": 0, | |
| "start_rank_id": 0, | |
| "strided_context_parallel_kernel_enabled": false, | |
| "switch_cc": false, | |
| "target": null, | |
| "tensor_capture_config": null, | |
| "tensor_replacement_config": null, | |
| "tile_cc": false, | |
| "tkg_batch_size": 4, | |
| "token_generation_batches": null, | |
| "token_generation_buckets": null, | |
| "token_tree_config": null, | |
| "torch_dtype": "bfloat16", | |
| "tp_degree": 4, | |
| "transpose_shared_experts_weights": false, | |
| "up_clamp_lower_limit": null, | |
| "up_clamp_upper_limit": null, | |
| "use_index_calc_kernel": false, | |
| "vocab_parallel": false, | |
| "weight_gather_seq_len_threshold": 32768, | |
| "weights_to_skip_layout_optimization": [], | |
| "windowed_context_encoding_size": null, | |
| "world_size": 4 | |
| }, | |
| "no_repeat_ngram_size": 0, | |
| "num_attention_heads": 48, | |
| "num_attention_heads_per_layer": [ | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64, | |
| 48, | |
| 64, | |
| 64, | |
| 64 | |
| ], | |
| "num_beam_groups": 1, | |
| "num_beams": 1, | |
| "num_cores_per_group": 1, | |
| "num_experts": 256, | |
| "num_experts_per_tok": 8, | |
| "num_hidden_layers": 40, | |
| "num_key_value_heads": 8, | |
| "num_local_experts": 256, | |
| "num_return_sequences": 1, | |
| "output_attentions": false, | |
| "output_hidden_states": false, | |
| "output_scores": false, | |
| "pad_token_id": 9, | |
| "partial_rotary_factor": 0.5, | |
| "prefix": null, | |
| "problem_type": null, | |
| "pruned_heads": {}, | |
| "remove_invalid_values": false, | |
| "repetition_penalty": 1.0, | |
| "return_dict": true, | |
| "return_dict_in_generate": false, | |
| "rms_norm_eps": 1e-06, | |
| "rope_parameters": { | |
| "full_attention": { | |
| "attention_factor": 1.0, | |
| "beta_fast": 64.0, | |
| "beta_slow": 1.0, | |
| "factor": 32.0, | |
| "original_max_position_embeddings": 4096, | |
| "partial_rotary_factor": 0.5, | |
| "rope_theta": 500000.0, | |
| "rope_type": "yarn" | |
| }, | |
| "original_max_position_embeddings": 4096, | |
| "rope_type": "yarn", | |
| "sliding_attention": { | |
| "partial_rotary_factor": 1.0, | |
| "rope_theta": 10000.0, | |
| "rope_type": "default" | |
| } | |
| }, | |
| "router_aux_loss_coef": 0.0, | |
| "sep_token_id": null, | |
| "shared_expert_intermediate_size": 512, | |
| "sliding_window": 512, | |
| "suppress_tokens": null, | |
| "task_specific_params": null, | |
| "temperature": 1.0, | |
| "tf_legacy_loss": false, | |
| "tie_encoder_decoder": false, | |
| "tie_word_embeddings": false, | |
| "tokenizer_class": null, | |
| "top_k": 50, | |
| "top_p": 1.0, | |
| "torchscript": false, | |
| "transformers_version": "4.57.6", | |
| "typical_p": 1.0, | |
| "use_bfloat16": false, | |
| "use_cache": true, | |
| "vocab_size": 100352 | |
| } |