K2-Horizon-32B-FP8 / config.json
LiqunMa's picture
Upload model
87d6535 verified
Raw
History Blame
3.33 kB
{
"architectures": [
"K2HorizonForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"attention_gate_func": null,
"auto_map": {
"AutoConfig": "configuration_k2_horizon.K2HorizonConfig",
"AutoModel": "modeling_k2_horizon.K2HorizonModel",
"AutoModelForCausalLM": "modeling_k2_horizon.K2HorizonForCausalLM"
},
"bos_token_id": 0,
"decoder_sparse_step": 1,
"dtype": "bfloat16",
"eos_token_id": 1,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 5120,
"initializer_range": 0.02,
"intermediate_size": 26624,
"layernorm_num_groups": 4,
"max_position_embeddings": 524288,
"mlp_only_layers": [
0,
1,
2,
3,
4,
5,
6,
7,
8,
9,
10,
11,
12,
13,
14,
15,
16,
17,
18,
19,
20,
21,
22,
23,
24,
25,
26,
27,
28,
29,
30,
31,
32,
33,
34,
35,
36,
37,
38,
39,
40,
41,
42,
43,
44,
45,
46,
47,
48,
49,
50,
51,
52,
53,
54,
55,
56,
57,
58,
59,
60,
61,
62,
63
],
"model_type": "k2_horizon",
"moe_gate_bias": false,
"moe_intermediate_size": 0,
"mova_num_experts": 0,
"mova_num_experts_per_tok": 0,
"norm_topk_prob": true,
"num_attention_heads": 64,
"num_experts": 0,
"num_experts_per_tok": 0,
"num_hidden_layers": 64,
"num_key_value_heads": 8,
"num_shared_experts": 0,
"output_router_logits": false,
"pad_token_id": null,
"quantization_config": {
"config_groups": {
"group_0": {
"format": "float-quantized",
"input_activations": {
"actorder": null,
"block_structure": null,
"dynamic": true,
"group_size": 128,
"num_bits": 8,
"observer": null,
"observer_kwargs": {},
"scale_dtype": null,
"strategy": "group",
"symmetric": true,
"type": "float",
"zp_dtype": null
},
"output_activations": null,
"targets": [
"Linear"
],
"weights": {
"actorder": null,
"block_structure": [
128,
128
],
"dynamic": false,
"group_size": null,
"num_bits": 8,
"observer": "memoryless_minmax",
"observer_kwargs": {},
"scale_dtype": null,
"strategy": "block",
"symmetric": true,
"type": "float",
"zp_dtype": null
}
}
},
"format": "float-quantized",
"global_compression_ratio": null,
"ignore": [
"lm_head"
],
"kv_cache_scheme": null,
"quant_method": "compressed-tensors",
"quantization_status": "compressed",
"sparsity_config": {},
"transform_config": {},
"version": "0.18.0"
},
"query_key_norm": false,
"rms_norm_eps": 1e-06,
"rope_head_dim": 128,
"rope_parameters": {
"rope_theta": 10000000.0,
"rope_type": "default"
},
"router_aux_loss_coef": 0.001,
"router_scaling_factor": 1.0,
"router_score_func": "sigmoid",
"sliding_window": null,
"tie_word_embeddings": false,
"transformers_version": "5.13.0",
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 250624
}