Download coding-50-nonuniform/config.json from ISTA-DASLab/Qwen3-Coder-Next-RCO-pruned: direct link, hf CLI and curl.
- Browser
- Download file 2.81 kB
-
https://huggingface.co/ISTA-DASLab/Qwen3-Coder-Next-RCO-pruned/resolve/main/coding-50-nonuniform/config.json
- Command line
-
hf download hf://ISTA-DASLab/Qwen3-Coder-Next-RCO-pruned/coding-50-nonuniform/config.json
-
curl -L -o config.json https://huggingface.co/ISTA-DASLab/Qwen3-Coder-Next-RCO-pruned/resolve/main/coding-50-nonuniform/config.json
2.81 kB
| { | |
| "architectures": [ | |
| "Qwen3NextForCausalLM" | |
| ], | |
| "attention_bias": false, | |
| "attention_dropout": 0, | |
| "bos_token_id": 151643, | |
| "decoder_sparse_step": 1, | |
| "dtype": "bfloat16", | |
| "eos_token_id": 151645, | |
| "full_attention_interval": 4, | |
| "head_dim": 256, | |
| "hidden_act": "silu", | |
| "hidden_size": 2048, | |
| "initializer_range": 0.02, | |
| "intermediate_size": 5120, | |
| "layer_types": [ | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "linear_attention", | |
| "full_attention" | |
| ], | |
| "linear_conv_kernel_dim": 4, | |
| "linear_key_head_dim": 128, | |
| "linear_num_key_heads": 16, | |
| "linear_num_value_heads": 32, | |
| "linear_value_head_dim": 128, | |
| "max_position_embeddings": 262144, | |
| "mlp_only_layers": [], | |
| "model_type": "qwen3_next", | |
| "moe_intermediate_size": 512, | |
| "norm_topk_prob": true, | |
| "num_attention_heads": 16, | |
| "num_experts": 512, | |
| "num_experts_per_tok": 10, | |
| "num_hidden_layers": 48, | |
| "num_key_value_heads": 2, | |
| "original_num_experts": 512, | |
| "output_router_logits": false, | |
| "partial_rotary_factor": 0.25, | |
| "per_layer_num_experts": [ | |
| 359, | |
| 279, | |
| 365, | |
| 194, | |
| 214, | |
| 270, | |
| 287, | |
| 272, | |
| 245, | |
| 212, | |
| 218, | |
| 231, | |
| 188, | |
| 213, | |
| 236, | |
| 272, | |
| 208, | |
| 281, | |
| 278, | |
| 265, | |
| 236, | |
| 227, | |
| 223, | |
| 250, | |
| 205, | |
| 230, | |
| 253, | |
| 288, | |
| 217, | |
| 289, | |
| 289, | |
| 281, | |
| 226, | |
| 213, | |
| 215, | |
| 238, | |
| 217, | |
| 226, | |
| 267, | |
| 303, | |
| 304, | |
| 297, | |
| 290, | |
| 285, | |
| 296, | |
| 269, | |
| 267, | |
| 300 | |
| ], | |
| "rms_norm_eps": 1e-06, | |
| "rope_scaling": null, | |
| "rope_theta": 5000000, | |
| "router_aux_loss_coef": 0.001, | |
| "shared_expert_intermediate_size": 512, | |
| "tie_word_embeddings": false, | |
| "transformers_version": "4.57.6", | |
| "use_cache": true, | |
| "use_sliding_window": false, | |
| "vocab_size": 151936 | |
| } | |