初始化项目,由ModelHub XC社区提供模型
Model: jiamingshan/AHA-L2A-Qwen3-1.7B-repro Source: Original Platform
This commit is contained in:
83
recipe/protocol.json
Normal file
83
recipe/protocol.json
Normal file
@@ -0,0 +1,83 @@
|
||||
{
|
||||
"release": "jiamingshan/AHA-L2A-Qwen3-1.7B-repro",
|
||||
"source_git": {
|
||||
"repository": "https://github.com/shanjiaming/AHA",
|
||||
"branch": "codex/l2a-router-granularity",
|
||||
"commit": "b47a549"
|
||||
},
|
||||
"base_checkpoint": {
|
||||
"location": "repository root",
|
||||
"description": "Qwen3-1.7B tuned vanilla, global step 6911",
|
||||
"sha256_model_safetensors": "22c971a65f7f1835b0e2a38b45f2f92191f5d8b63b30969f2f61a69d30afbc05"
|
||||
},
|
||||
"execution": {
|
||||
"distributed_training": false,
|
||||
"world_size_per_arm": 1,
|
||||
"gpu_count_per_arm": 1,
|
||||
"one_gpu_machine": "run AHA then L2A-style sequentially on GPU 0",
|
||||
"two_or_more_gpu_machine": "run AHA on GPU 0 and L2A-style on GPU 1 concurrently; remaining GPUs are unused",
|
||||
"arm_definition": "one independent experimental variant, not a processor architecture or GPU group"
|
||||
},
|
||||
"arms": {
|
||||
"aha": {
|
||||
"router_granularity": "token_kv_head",
|
||||
"description": "one native gate per token, KV head, and layer"
|
||||
},
|
||||
"l2a_style": {
|
||||
"router_granularity": "token",
|
||||
"description": "one native gate per token and layer, shared across all heads",
|
||||
"claim_boundary": "L2A-style shared-gate; not an official L2A reproduction"
|
||||
}
|
||||
},
|
||||
"common": {
|
||||
"local_attention": {
|
||||
"kind": "sink_recent",
|
||||
"sink_tokens": 64,
|
||||
"recent_tokens": 256
|
||||
},
|
||||
"gate_initialization": {
|
||||
"weight": "zeros",
|
||||
"full_probability": 0.9
|
||||
},
|
||||
"effective_sparsity_denominator": "token x KV-head x layer"
|
||||
},
|
||||
"stage_1": {
|
||||
"dataset": "recipe/data/am_distilled_long_mix",
|
||||
"rows": 1024,
|
||||
"max_length": 8192,
|
||||
"batch_size": 1,
|
||||
"optimizer_steps": 300,
|
||||
"epochs": 0.29296875,
|
||||
"trainable": "native gate rows only",
|
||||
"learning_rate": 0.00003,
|
||||
"warmup_ratio": 0.1,
|
||||
"hidden_state_distillation": 1.0,
|
||||
"regularizer": 0.1,
|
||||
"ce": 0.0,
|
||||
"train_threshold": 0.5,
|
||||
"seed": 42
|
||||
},
|
||||
"stage_2": {
|
||||
"dataset": "same 1024 rows",
|
||||
"max_length": 8192,
|
||||
"batch_size": 1,
|
||||
"optimizer_steps_run": 75,
|
||||
"selected_checkpoint_step": 25,
|
||||
"selected_epochs": 0.0244140625,
|
||||
"gate_learning_rate": 0.000003,
|
||||
"backbone_learning_rate": 0.0000003,
|
||||
"ce": 1.0,
|
||||
"attention_distillation": 0.5,
|
||||
"regularizer": 0.01,
|
||||
"train_threshold": 0.58,
|
||||
"freeze_embeddings_and_lm_head": true,
|
||||
"seed": 47
|
||||
},
|
||||
"evaluation": {
|
||||
"thresholds": [0.45, 0.5, 0.525, 0.55, 0.575, 0.6, 0.625, 0.65],
|
||||
"strict_prefill_and_decode_sparse_routing": true,
|
||||
"force_full_heads": false,
|
||||
"full_decode": false,
|
||||
"headline": "highest measured effective sparsity whose score is at least 95% of tuned vanilla on every suite"
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user