{ "smoke": false, "cuda": false, "backend": "cpu-host", "architecture": "gpt2-scratch", "param_count": 443719680, "n_layer": 22, "n_embd": 1280, "n_head": 20, "vocab_size": 8192, "max_seq_len": 256, "steps": 120, "requested_steps": 120, "gradient_checkpointing": true, "gradient_accumulation": 2, "batch_size": 1, "learning_rate": 0.0003, "rows": 3175, "dataset_version": "1.4.0", "final_loss": 5.811557769775391, "losses_tail": [ 6.735813140869141, 5.894803047180176, 5.694976806640625, 6.798207759857178, 6.259979724884033, 5.799972057342529, 6.757445335388184, 6.014622211456299, 6.114091396331787, 5.811557769775391 ], "hub_id_if_uploaded": "theworker02/open-reason-xl", "card_title": "Open Reason open-reason-xl (CPU)", "size_note": "This is an **XL** GPT-2-style causal LM (~450M parameters) trained from scratch on the Open Reason SFT split. It is larger than `theworker02/open-reason-large` and is **not** a 1B model and is **not** `theworker02/open-reason-1b`.", "hardware": "Host CPU (AMD64 Family 26 Model 68 Stepping 0, AuthenticAMD); torch 2.12.0+cpu; cuda_available=False; docker_installed=False; docker_used=False. NVIDIA CUDA was not used. AMD GPU/ROCm/DirectML were not used.", "docker_used": false, "docker_installed": false, "torch_version": "2.12.0+cpu", "note": "CPU causal LM. Not open-reason-1b. Not AMD GPU. No Reddit." }