54 lines
1.3 KiB
JSON
54 lines
1.3 KiB
JSON
{
|
|
"config": {
|
|
"model_name": "HuggingFaceTB/SmolLM2-135M-Instruct",
|
|
"dataset_name": "RomanTeucher/text2cypher-curated",
|
|
"output_dir": "outputs/smollm2-135m-text2cypher",
|
|
"max_length": 1024,
|
|
"epochs": 20.0,
|
|
"learning_rate": 5e-05,
|
|
"weight_decay": 0.01,
|
|
"warmup_ratio": 0.05,
|
|
"warmup_steps": null,
|
|
"lr_scheduler_type": "cosine",
|
|
"train_batch_size": 4,
|
|
"eval_batch_size": 4,
|
|
"gradient_accumulation_steps": 4,
|
|
"max_grad_norm": 1.0,
|
|
"bf16": null,
|
|
"fp16": null,
|
|
"use_lora": true,
|
|
"lora_r": 16,
|
|
"lora_alpha": 32,
|
|
"lora_dropout": 0.05,
|
|
"logging_steps": 10,
|
|
"eval_strategy": "epoch",
|
|
"save_strategy": "epoch",
|
|
"save_total_limit": 1,
|
|
"load_best_model_at_end": true,
|
|
"metric_for_best_model": "eval_loss",
|
|
"greater_is_better": false,
|
|
"seed": 42,
|
|
"cpu_threads": null,
|
|
"push_to_hub": false,
|
|
"hub_model_id": null
|
|
},
|
|
"tokenization": {
|
|
"train": {
|
|
"n": 1000,
|
|
"truncated": 0,
|
|
"empty_target": 0
|
|
},
|
|
"val": {
|
|
"n": 75,
|
|
"truncated": 0,
|
|
"empty_target": 0
|
|
}
|
|
},
|
|
"final_eval": {
|
|
"eval_loss": 0.3248527944087982,
|
|
"eval_runtime": 2.1616,
|
|
"eval_samples_per_second": 34.696,
|
|
"eval_steps_per_second": 8.79,
|
|
"epoch": 20.0
|
|
}
|
|
} |