Files
babylm-lem-spa-ell-sequenti…/training_metadata.json

71 lines
931 B
JSON
Raw Normal View History

{
"input_config": {
"model_type": "gpt2",
"architectures": [
"GPT2LMHeadModel"
],
"activation_function": "gelu",
"attn_pdrop": 0.1,
"embd_pdrop": 0.1,
"resid_pdrop": 0.1,
"initializer_range": 0.02,
"layer_norm_epsilon": 1e-05,
"n_embd": 768,
"n_inner": 3072,
"n_head": 12,
"n_layer": 12,
"vocab_size": 1,
"n_ctx": 512,
"n_positions": 512
},
"num_steps_switch": 8720,
"save_steps": [
1,
2,
4,
8,
16,
32,
64,
128,
256,
512,
1000,
2000,
3000,
4000,
5000,
6000,
7000,
8000,
8720,
8721,
8722,
8724,
8728,
8736,
8752,
8784,
8848,
8976,
9000,
9232,
10000,
11000,
12000,
13000,
14000,
15000,
16000,
17000,
18000,
19000,
20000,
21000,
22000,
23000,
24000,
25000,
25460
]
}