- Restructure repo around experiments/<name>/ and platforms/<chip>.env. - Add shared scripts under scripts/common/ for platform/server/bench-client logic. - Add Kunlun P800 platform config and runtime patches. - Add dsv4_p800_sglang experiment with INT8 smoke-test support. - Update BENCHMARK_WORKFLOW.md and README.md with chip/engine recording rules. - Add scripts/analysis/compare_experiments.py for cross-experiment comparison. - Ignore experiments/*/results/ raw output directories by default.
114 lines
1.9 KiB
JSON
114 lines
1.9 KiB
JSON
{
|
|
"architectures": [
|
|
"DeepseekXYZForCausalLM"
|
|
],
|
|
"attention_bias": false,
|
|
"attention_dropout": 0.0,
|
|
"bos_token_id": 0,
|
|
"eos_token_id": 1,
|
|
"hc_eps": 1e-06,
|
|
"hc_mult": 4,
|
|
"hc_sinkhorn_iters": 20,
|
|
"head_dim": 512,
|
|
"hidden_act": "silu",
|
|
"hidden_size": 4096,
|
|
"index_head_dim": 128,
|
|
"index_n_heads": 64,
|
|
"index_topk": 512,
|
|
"initializer_range": 0.02,
|
|
"max_position_embeddings": 1048576,
|
|
"model_type": "deepseek_ref",
|
|
"moe_intermediate_size": 2048,
|
|
"n_routed_experts": 256,
|
|
"n_shared_experts": 1,
|
|
"norm_topk_prob": true,
|
|
"num_attention_heads": 64,
|
|
"num_experts_per_tok": 6,
|
|
"num_hidden_layers": 43,
|
|
"num_hash_layers": 3,
|
|
"num_key_value_heads": 1,
|
|
"num_nextn_predict_layers": 1,
|
|
"o_groups": 8,
|
|
"o_lora_rank": 1024,
|
|
"q_lora_rank": 1024,
|
|
"qk_rope_head_dim": 64,
|
|
"rms_norm_eps": 1e-06,
|
|
"rope_scaling": {
|
|
"beta_fast": 32.0,
|
|
"beta_slow": 1.0,
|
|
"factor": 16.0,
|
|
"original_max_position_embeddings": 65536,
|
|
"type": "yarn"
|
|
},
|
|
"rope_theta": 10000,
|
|
"routed_scaling_factor": 1.5,
|
|
"scoring_func": "sqrtsoftplus",
|
|
"sliding_window": 128,
|
|
"swiglu_limit": 10.0,
|
|
"tie_word_embeddings": false,
|
|
"n_group": 8,
|
|
"topk_group": 8,
|
|
"topk_method": "noaux_tc",
|
|
"torch_dtype": "bfloat16",
|
|
"transformers_version": "4.57.1",
|
|
"use_cache": true,
|
|
"vocab_size": 129280,
|
|
"compress_rope_theta": 160000,
|
|
"compress_ratios": [
|
|
0,
|
|
0,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
128,
|
|
4,
|
|
0
|
|
],
|
|
"quantization_config": {
|
|
"activation_scheme": "dynamic",
|
|
"fmt": "e4m3",
|
|
"quant_method": "fp8",
|
|
"scale_fmt": "ue8m0",
|
|
"weight_block_size": [
|
|
128,
|
|
128
|
|
]
|
|
},
|
|
"expert_dtype": "fp4"
|
|
} |