open-reason-xl / train_metrics.json
theworker02's picture
Upload Open Reason CPU causal LM theworker02/open-reason-xl (not 1B)
9b47349 verified
Raw
History Blame Contribute Delete
1.49 kB
{
"smoke": false,
"cuda": false,
"backend": "cpu-host",
"architecture": "gpt2-scratch",
"param_count": 443719680,
"n_layer": 22,
"n_embd": 1280,
"n_head": 20,
"vocab_size": 8192,
"max_seq_len": 256,
"steps": 120,
"requested_steps": 120,
"gradient_checkpointing": true,
"gradient_accumulation": 2,
"batch_size": 1,
"learning_rate": 0.0003,
"rows": 3175,
"dataset_version": "1.4.0",
"final_loss": 5.811557769775391,
"losses_tail": [
6.735813140869141,
5.894803047180176,
5.694976806640625,
6.798207759857178,
6.259979724884033,
5.799972057342529,
6.757445335388184,
6.014622211456299,
6.114091396331787,
5.811557769775391
],
"hub_id_if_uploaded": "theworker02/open-reason-xl",
"card_title": "Open Reason open-reason-xl (CPU)",
"size_note": "This is an **XL** GPT-2-style causal LM (~450M parameters) trained from scratch on the Open Reason SFT split. It is larger than `theworker02/open-reason-large` and is **not** a 1B model and is **not** `theworker02/open-reason-1b`.",
"hardware": "Host CPU (AMD64 Family 26 Model 68 Stepping 0, AuthenticAMD); torch 2.12.0+cpu; cuda_available=False; docker_installed=False; docker_used=False. NVIDIA CUDA was not used. AMD GPU/ROCm/DirectML were not used.",
"docker_used": false,
"docker_installed": false,
"torch_version": "2.12.0+cpu",
"note": "CPU causal LM. Not open-reason-1b. Not AMD GPU. No Reddit."
}