{ "run_info": { "created_at": "2026-08-05T19:21:57+00:00", "total_time": 4876.834152824013, "experiment_name": "lokr/llama-3.2-3B-rank32-factor4-all-linear-lr0.005", "peft_branch": "auto-research", "train_config": { "model_id": "meta-llama/Llama-3.2-3B", "dtype": "bfloat16", "max_seq_length": 768, "batch_size": 4, "batch_size_eval": 50, "max_steps": 5000, "eval_steps": 250, "compile": false, "use_gc": false, "query_template": "Question: {query} Think step by step.\nAnswer:", "seed": 0, "grad_norm_clip": 1.0, "optimizer_type": "AdamW", "optimizer_kwargs": { "lr": 0.005, "weight_decay": 0.1 }, "lr_scheduler": "cosine", "use_amp": false, "autocast_adapter_dtype": true, "generation_kwargs": { "max_length": 800, "max_new_tokens": 300 }, "attn_implementation": null, "init_kv_cache_prefix": null }, "peft_config": { "task_type": null, "peft_type": "LOKR", "auto_mapping": null, "peft_version": "0.20.1.dev0@UNKNOWN", "base_model_name_or_path": "meta-llama/Llama-3.2-3B", "revision": null, "inference_mode": false, "rank_pattern": {}, "alpha_pattern": {}, "r": 32, "alpha": 64, "rank_dropout": 0.0, "module_dropout": 0.0, "use_effective_conv2d": false, "decompose_both": false, "decompose_factor": 4, "rank_dropout_scale": false, "target_modules": [ "up_proj", "gate_proj", "v_proj", "q_proj", "k_proj", "o_proj", "down_proj" ], "exclude_modules": null, "init_weights": true, "layers_to_transform": null, "layers_pattern": null, "modules_to_save": null }, "error_msg": "" }, "train_info": { "accelerator_memory_reserved_avg": 29428636922, "accelerator_memory_max": 40980447232, "accelerator_memory_reserved_99th": 37629198336, "train_time": 4337.368219144992, "file_size": 48714440, "num_trainable_params": 12160064, "num_total_params": 3224909888, "status": "success", "metrics": [ { "step": 250, "valid accuracy": 0.4, "train loss": 0.8243188563585281, "train samples": 1000, "train time": 173.05660409928532, "eval time": 30.925348686025245, "tokens / sec": 1223.4089597559273, "mem allocated avg": 6974068426.752, "mem reserved avg": 29637765758.976, "elapsed time": 234.67427716800012 }, { "step": 500, "valid accuracy": 0.32, "train loss": 0.7367534626722336, "train samples": 2000, "train time": 168.92729432377382, "eval time": 19.947831519006286, "tokens / sec": 1231.2693507145577, "mem allocated avg": 6965838184.448, "mem reserved avg": 29036730384.384, "elapsed time": 431.30717777099926 }, { "step": 750, "valid accuracy": 0.22, "train loss": 0.7591401909589768, "train samples": 3000, "train time": 173.5782248750329, "eval time": 30.72412709100172, "tokens / sec": 1235.1837343327904, "mem allocated avg": 6976710492.16, "mem reserved avg": 29298144575.488, "elapsed time": 643.3134106380166 }, { "step": 1000, "valid accuracy": 0.16, "train loss": 0.7286917352676392, "train samples": 4000, "train time": 173.25569501990685, "eval time": 15.424150008999277, "tokens / sec": 1202.4770670658904, "mem allocated avg": 6968188157.952, "mem reserved avg": 29486888255.488, "elapsed time": 839.8342328320141 }, { "step": 1250, "valid accuracy": 0.24, "train loss": 0.7228857347965241, "train samples": 5000, "train time": 171.9056585111539, "eval time": 30.715734964003786, "tokens / sec": 1213.0956118961567, "mem allocated avg": 6967430975.488, "mem reserved avg": 29512901328.896, "elapsed time": 1050.2152808600222 }, { "step": 1500, "valid accuracy": 0.28, "train loss": 0.7035419543981553, "train samples": 6000, "train time": 171.0979807379772, "eval time": 30.68885849401704, "tokens / sec": 1223.4568701343917, "mem allocated avg": 6969443014.656, "mem reserved avg": 29386233348.096, "elapsed time": 1259.8666626070044 }, { "step": 1750, "valid accuracy": 0.38, "train loss": 0.6936728912591934, "train samples": 7000, "train time": 172.45823883989942, "eval time": 30.67647124498035, "tokens / sec": 1213.9460625847714, "mem allocated avg": 6970556262.4, "mem reserved avg": 29963210194.944, "elapsed time": 1470.7583586190012 }, { "step": 2000, "valid accuracy": 0.34, "train loss": 0.6850932078361511, "train samples": 8000, "train time": 172.2125422480749, "eval time": 21.635232597007416, "tokens / sec": 1206.044561497795, "mem allocated avg": 6966487302.144, "mem reserved avg": 29441472331.776, "elapsed time": 1672.5279613650055 }, { "step": 2250, "valid accuracy": 0.38, "train loss": 0.6678359975814819, "train samples": 9000, "train time": 174.90155081104604, "eval time": 30.710825373011176, "tokens / sec": 1228.9656609861506, "mem allocated avg": 6978265659.392, "mem reserved avg": 29786906820.608, "elapsed time": 1885.9478800620127 }, { "step": 2500, "valid accuracy": 0.44, "train loss": 0.6567883492708206, "train samples": 10000, "train time": 168.23802922523464, "eval time": 30.644450194988167, "tokens / sec": 1224.2594670688538, "mem allocated avg": 6963658600.448, "mem reserved avg": 29001515008.0, "elapsed time": 2092.6179425250157 }, { "step": 2750, "valid accuracy": 0.46, "train loss": 0.634885214805603, "train samples": 11000, "train time": 173.90507797230384, "eval time": 30.730780140991556, "tokens / sec": 1218.371553438734, "mem allocated avg": 6973873780.736, "mem reserved avg": 29700227334.144, "elapsed time": 2304.930961822014 }, { "step": 3000, "valid accuracy": 0.5, "train loss": 0.6183034368753433, "train samples": 12000, "train time": 173.86735759672592, "eval time": 30.715003296005307, "tokens / sec": 1200.5186188205496, "mem allocated avg": 6969537423.36, "mem reserved avg": 29600050577.408, "elapsed time": 2517.428843543021 }, { "step": 3250, "valid accuracy": 0.48, "train loss": 0.6171810357570648, "train samples": 13000, "train time": 172.11153198982356, "eval time": 20.41151063601137, "tokens / sec": 1225.3740209137754, "mem allocated avg": 6971543304.192, "mem reserved avg": 29378851373.056, "elapsed time": 2717.537005299004 }, { "step": 3500, "valid accuracy": 0.54, "train loss": 0.5918659865856171, "train samples": 14000, "train time": 171.7280791127414, "eval time": 20.10971114897984, "tokens / sec": 1221.407710863037, "mem allocated avg": 6969227558.912, "mem reserved avg": 29302649257.984, "elapsed time": 2917.0929282610014 }, { "step": 3750, "valid accuracy": 0.52, "train loss": 0.5810757111310959, "train samples": 15000, "train time": 177.90301188721787, "eval time": 30.743899679015158, "tokens / sec": 1218.09629697208, "mem allocated avg": 6980796526.592, "mem reserved avg": 29780892188.672, "elapsed time": 3133.516127213021 }, { "step": 4000, "valid accuracy": 0.46, "train loss": 0.5831188062429428, "train samples": 16000, "train time": 169.06811870416277, "eval time": 30.76702547399327, "tokens / sec": 1208.8204539474061, "mem allocated avg": 6962208227.328, "mem reserved avg": 29357410091.008, "elapsed time": 3341.2088356550084 }, { "step": 4250, "valid accuracy": 0.54, "train loss": 0.564371225476265, "train samples": 17000, "train time": 173.65914930703002, "eval time": 30.680932629999006, "tokens / sec": 1217.2638230898128, "mem allocated avg": 6971919788.032, "mem reserved avg": 29329702518.784, "elapsed time": 3553.1107871610147 }, { "step": 4500, "valid accuracy": 0.54, "train loss": 0.5646625064611435, "train samples": 18000, "train time": 168.5221932909335, "eval time": 21.17475585200009, "tokens / sec": 1233.1788231667917, "mem allocated avg": 6967363680.256, "mem reserved avg": 29285041569.792, "elapsed time": 3750.635109158 }, { "step": 4750, "valid accuracy": 0.5, "train loss": 0.5495809932947159, "train samples": 19000, "train time": 171.79509386298014, "eval time": 20.51026588899549, "tokens / sec": 1222.0314054337464, "mem allocated avg": 6969873039.36, "mem reserved avg": 29383817428.992, "elapsed time": 3950.5909262890054 }, { "step": 5000, "valid accuracy": 0.54, "train loss": 0.5584390600919723, "train samples": 20000, "train time": 170.36222512708628, "eval time": 27.515913671988528, "tokens / sec": 1222.5714934436196, "mem allocated avg": 6966943275.008, "mem reserved avg": 28902328107.008, "elapsed time": 4156.128801064013 }, { "step": 5000, "test accuracy": 0.5064442759666414, "train loss": 0.5584390600919723, "train samples": 20000, "train total tokens": 4198051, "forgetting": 0.15337693691253662 } ] }, "meta_info": { "model_info": { "sha": "13afe5124825b4f3751f836b40dafda64c1ed062", "created_at": "2024-09-18T15:23:48+00:00" }, "dataset_info": { "metamath": { "sha": "aa4f34d3d2d3231299b5b03d9b3e5a20da45aa18", "created_at": "2023-09-21T17:22:46+00:00" }, "gsm8k": { "sha": "740312add88f781978c0658806c59bc2815b9866", "created_at": "2022-04-12T10:22:10+00:00" } }, "package_info": { "transformers-version": "5.12.1", "transformers-commit-hash": null, "peft-version": "0.20.1.dev0", "peft-commit-hash": "c702de402330f48c0d8b1f260d59413aa3110b65", "datasets-version": "4.8.4", "datasets-commit-hash": null, "bitsandbytes-version": "0.49.2", "bitsandbytes-commit-hash": null, "torch-version": "2.11.0+cu130", "torch-commit-hash": null }, "system_info": { "system": "Linux", "release": "7.0.0-1009-aws", "version": "#9~24.04.1-Ubuntu SMP PREEMPT Mon Jul 13 22:43:41 UTC 2026", "machine": "x86_64", "processor": "x86_64", "accelerator": "NVIDIA L40S" }, "pytorch_info": "PyTorch built with:\n - GCC 13.3\n - C++ Version: 201703\n - Intel(R) oneAPI Math Kernel Library Version 2024.2-Product Build 20240605 for Intel(R) 64 architecture applications\n - Intel(R) MKL-DNN v3.10.2 (Git Hash f1d471933dc852f956fd05389f9313c7148783d5)\n - OpenMP 201511 (a.k.a. OpenMP 4.5)\n - LAPACK is enabled (usually provided by MKL)\n - NNPACK is enabled\n - CPU capability usage: AVX2\n - CUDA Runtime 13.0\n - NVCC architecture flags: -gencode;arch=compute_75,code=sm_75;-gencode;arch=compute_80,code=sm_80;-gencode;arch=compute_86,code=sm_86;-gencode;arch=compute_90,code=sm_90;-gencode;arch=compute_100,code=sm_100;-gencode;arch=compute_120,code=sm_120\n - CuDNN 90.7.1 (built against CUDA 12.8)\n - Built with CuDNN 91.9\n - Magma 2.6.1\n - Build settings: BLAS_INFO=mkl, BUILD_TYPE=Release, COMMIT_SHA=70d99e998b4955e0049d13a98d77ae1b14db1f45, CUDA_VERSION=13.0, CUDNN_VERSION=9.19.0, CXX_COMPILER=/opt/rh/gcc-toolset-13/root/usr/bin/c++, CXX_FLAGS= -fvisibility-inlines-hidden -DUSE_PTHREADPOOL -DNDEBUG -DUSE_KINETO -DLIBKINETO_NOROCTRACER -DLIBKINETO_NOXPUPTI=ON -DUSE_FBGEMM -DUSE_MSLK -DUSE_PYTORCH_QNNPACK -DUSE_XNNPACK -DSYMBOLICATE_MOBILE_DEBUG_HANDLE -O2 -fPIC -DC10_NODEPRECATED -Wall -Wextra -Werror=return-type -Werror=non-virtual-dtor -Werror=range-loop-construct -Werror=bool-operation -Wnarrowing -Wno-missing-field-initializers -Wno-unknown-pragmas -Wno-unused-parameter -Wno-strict-overflow -Wno-strict-aliasing -Wno-stringop-overflow -Wsuggest-override -Wno-psabi -Wno-error=old-style-cast -faligned-new -Wno-maybe-uninitialized -fno-math-errno -fno-trapping-math -Werror=format -Wno-dangling-reference -Wno-error=dangling-reference -Wno-stringop-overflow, LAPACK_INFO=mkl, PERF_WITH_AVX=1, PERF_WITH_AVX2=1, TORCH_VERSION=2.11.0, USE_CUDA=ON, USE_CUDNN=ON, USE_CUSPARSELT=1, USE_GFLAGS=OFF, USE_GLOG=OFF, USE_GLOO=ON, USE_MKL=ON, USE_MKLDNN=ON, USE_MPI=OFF, USE_NCCL=1, USE_NNPACK=ON, USE_OPENMP=ON, USE_ROCM=OFF, USE_ROCM_KERNEL_ASSERT=OFF, USE_XCCL=OFF, USE_XPU=OFF, \n" } }