Files
smollm2-135m-text2cypher/train_summary.json

54 lines
1.3 KiB
JSON
Raw Permalink Normal View History

{
"config": {
"model_name": "HuggingFaceTB/SmolLM2-135M-Instruct",
"dataset_name": "RomanTeucher/text2cypher-curated",
"output_dir": "outputs/smollm2-135m-text2cypher",
"max_length": 1024,
"epochs": 20.0,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.05,
"warmup_steps": null,
"lr_scheduler_type": "cosine",
"train_batch_size": 4,
"eval_batch_size": 4,
"gradient_accumulation_steps": 4,
"max_grad_norm": 1.0,
"bf16": null,
"fp16": null,
"use_lora": true,
"lora_r": 16,
"lora_alpha": 32,
"lora_dropout": 0.05,
"logging_steps": 10,
"eval_strategy": "epoch",
"save_strategy": "epoch",
"save_total_limit": 1,
"load_best_model_at_end": true,
"metric_for_best_model": "eval_loss",
"greater_is_better": false,
"seed": 42,
"cpu_threads": null,
"push_to_hub": false,
"hub_model_id": null
},
"tokenization": {
"train": {
"n": 1000,
"truncated": 0,
"empty_target": 0
},
"val": {
"n": 75,
"truncated": 0,
"empty_target": 0
}
},
"final_eval": {
"eval_loss": 0.3248527944087982,
"eval_runtime": 2.1616,
"eval_samples_per_second": 34.696,
"eval_steps_per_second": 8.79,
"epoch": 20.0
}
}