{ "created_at": "2026-07-16 23:52:50", "name": "v52_public091_full_lr2e5_ep1", "base": "OpenOneRec/OneReason-0.8B-pretrain-competition", "dataset": "v49_public_091_exact", "dataset_manifest": { "path": "/home/ll/llm4rec/experiments/data/v49_public_091_exact.jsonl", "records": 32705, "groups": { "rec": 18651, "item": 9684, "user": 2792, "world": 1578 }, "routes": { "no_think": 20934, "think": 11771 }, "thought_modes": { "empty": 20934, "filled": 11771 }, "sha256": "cdedea13c560d3453697f4a3c9b96a2303bd708c84833034fa953d16482b9925", "source_path": "/home/ll/llm4rec/data/external/kuaishou-llmrec-sft-baseline-0.91/train.jsonl", "source_url": "https://huggingface.co/datasets/Frinkleko/kuaishou-llmrec-sft-baseline-0.91/resolve/main/train.jsonl?download=true", "source_sha256": "4d6b29d76974c9a1517c1b583858e744cae019cb26e1e2d90066e237ebbcf5f8", "source_license": "Apache-2.0", "transform": "Lossless field mapping from system/prompt/response to instruction/input/output; original seed-42 order retained." }, "learning_rate": "2.0e-5", "epochs": 1.0, "method": "full", "weight_decay": 0.0, "validation_size": 0.0, "lora": { "rank": null, "alpha": null, "dropout": null, "target": null }, "stages": null, "note": "Full-SFT counterpart using v7's successful 2e-5 scale, now on balanced CoT/direct clean data instead of final-only data.", "evaluation": null, "cot_policy": "Preserve meaningful CoT for /think and pure final outputs for /no_think.", "checkpoint_policy": "save_strategy=no; only the final model is retained.", "script": "/home/ll/llm4rec/experiments/run_experiments.py", "script_sha256": "bc57a4d1337c7182996ece63ddc844185043b68ae2a2dbe793e3969bbccc2e2d", "data_builder": "/home/ll/llm4rec/experiments/source_data.py", "data_builder_sha256": "a1216b14c8c516f9fde2fcaab9d70ec5b8a200033fbd1ee330c6b461213587ad", "data_cleaner": null, "data_cleaner_sha256": null, "configs": [ "/home/ll/llm4rec/experiments/configs/v52_public091_full_lr2e5_ep1.yaml" ], "reproduce": "/home/ll/llm4rec/demo/LLaMA-Factory/.venv/bin/python /home/ll/llm4rec/experiments/run_experiments.py --single v52_public091_full_lr2e5_ep1 --gpu ", "training_result": { "name": "v52_public091_full_lr2e5_ep1", "gpu": 1, "started_at": "2026-07-17 01:41:03", "status": "ok", "wall_seconds": 2611.22, "finished_at": "2026-07-17 02:24:34", "epoch": 1.0, "total_flos": 1.563295846805975e+17, "train_loss": 1.6355595296145948, "train_runtime": 2594.717, "train_samples_per_second": 0.503, "train_steps_per_second": 0.126, "global_step": 326, "first_logged_loss": 2.659077453613281, "last_logged_loss": 1.512490463256836, "min_logged_loss": 1.5008227348327636, "tail_loss_mean": 1.5448276201883953, "logged_loss_change": -1.1465869903564452, "batch": "v49-v53" } }