CoolFace
Datasetpublic

kobzaond/RLVRAMBench

RLVRAMBench Which language-model training configurations can I use with the memory I have, and how much testing does that decision require? RLVRAMBench is a measurement dataset with open evaluation tasks for a specific language-model training system. It measures memory feasibility when response generation and reinforcement-learning updates share the same graphics processors. It provides measured outcomes, fixed prediction tasks, a budgeted decision replay, reference methods, and… See the full description on the dataset page: https://huggingface.co/datasets/kobzaond/RLVRAMBench.

sourceHugging Facemitupdated 8d agoView on Hugging Face
0likes223downloads
model_metadata.json109 linesDownload Raw Back to estimation
1{2  "Qwen/Qwen2.5-3B-Instruct": {3    "all_linear_layer_adapter_elements": 119734272,4    "config_sha256": "eed00b17e22553979d090fa492e587e92885e328914c8e0b0b78f0a0d3576b3b",5    "floating_checkpoint_bytes": 6171877376,6    "floating_checkpoint_elements": 3085938688,7    "head_dim": 128,8    "hidden_size": 2048,9    "intermediate_size": 11008,10    "largest_transformer_layer_elements": 77076992,11    "largest_weight_tensor_elements": 311164928,12    "lora_rank": 64,13    "model_id": "Qwen/Qwen2.5-3B-Instruct",14    "non_transformer_layer_elements": 311166976,15    "num_attention_heads": 16,16    "num_hidden_layers": 36,17    "num_key_value_heads": 2,18    "revision": "aa8e72537993ba99e69dfaafa59ed015b17504d1",19    "scope": "Numeric architecture/checkpoint-header inputs only; no weight payload. Adapter count assumes rank-64 all-linear transformer layers, excluding embeddings/output head. Floating checkpoint elements may include buffers; no separately allocated reference model is assumed.",20    "shard_header_sha256": {21      "model-00001-of-00002.safetensors": "8d1c57f12799cf1a917776014a04eff3bfaf1cb1dafa2ca3a12f92ade34e65bd",22      "model-00002-of-00002.safetensors": "5bb7481ec0a898ebaa815dbdd881d83752dfc05e445e19a7571800dbd025cf8e"23    },24    "tensor_count": 434,25    "tie_word_embeddings": true,26    "vocab_size": 15193627  },28  "Qwen/Qwen2.5-7B-Instruct": {29    "all_linear_layer_adapter_elements": 161480704,30    "config_sha256": "7463bb0ea78315365e6c6b74de4e73bbcc8359dfb0c5a737584e077d42c0b03c",31    "floating_checkpoint_bytes": 15231233024,32    "floating_checkpoint_elements": 7615616512,33    "head_dim": 128,34    "hidden_size": 3584,35    "intermediate_size": 18944,36    "largest_transformer_layer_elements": 233057792,37    "largest_weight_tensor_elements": 544997376,38    "lora_rank": 64,39    "model_id": "Qwen/Qwen2.5-7B-Instruct",40    "non_transformer_layer_elements": 1089998336,41    "num_attention_heads": 28,42    "num_hidden_layers": 28,43    "num_key_value_heads": 4,44    "revision": "a09a35458c702b33eeacc393d103063234e8bc28",45    "scope": "Numeric architecture/checkpoint-header inputs only; no weight payload. Adapter count assumes rank-64 all-linear transformer layers, excluding embeddings/output head. Floating checkpoint elements may include buffers; no separately allocated reference model is assumed.",46    "shard_header_sha256": {47      "model-00001-of-00004.safetensors": "24ce65896dba011b065dd8fcbf877559a6701fe80516f7d4d039d082147cb2e8",48      "model-00002-of-00004.safetensors": "69f0c2c87ae06b271181d6964cf235aa9ce7931f64cde7a1a5511ba16070c739",49      "model-00003-of-00004.safetensors": "a859e7527f6cf5f144667f65c356c01374f95ed56f40cbe6dca4cb7752164066",50      "model-00004-of-00004.safetensors": "181209f962ac9f71cb2bf72adbd5727cf4fbbd646cf80558576ccb5c13e75b71"51    },52    "tensor_count": 339,53    "tie_word_embeddings": false,54    "vocab_size": 15206455  },56  "ibm-granite/granite-3.3-2b-instruct": {57    "all_linear_layer_adapter_elements": 112721920,58    "config_sha256": "9202d328d8368958aab7dad89a8e6aa35c250b3a2eedd2d8850ee0bcceca65f6",59    "floating_checkpoint_bytes": 5067079680,60    "floating_checkpoint_elements": 2533539840,61    "head_dim": 64,62    "hidden_size": 2048,63    "intermediate_size": 8192,64    "largest_transformer_layer_elements": 60821504,65    "largest_weight_tensor_elements": 100677632,66    "lora_rank": 64,67    "model_id": "ibm-granite/granite-3.3-2b-instruct",68    "non_transformer_layer_elements": 100679680,69    "num_attention_heads": 32,70    "num_hidden_layers": 40,71    "num_key_value_heads": 8,72    "revision": "707f574c62054322f6b5b04b6d075f0a8f05e0f0",73    "scope": "Numeric architecture/checkpoint-header inputs only; no weight payload. Adapter count assumes rank-64 all-linear transformer layers, excluding embeddings/output head. Floating checkpoint elements may include buffers; no separately allocated reference model is assumed.",74    "shard_header_sha256": {75      "model-00001-of-00002.safetensors": "744c520c1a1d794563fed804e54d99dd424713d4346c3edc06dfdd17c6afe1c0",76      "model-00002-of-00002.safetensors": "039f729e7d783e8514dcac3af74ef1e9ad7c8c235e7d92671000e778ab937d79"77    },78    "tensor_count": 362,79    "tie_word_embeddings": true,80    "vocab_size": 4915981  },82  "microsoft/Phi-4-mini-instruct": {83    "all_linear_layer_adapter_elements": 92274688,84    "config_sha256": "ac65d86061d3d0d704ee2511fd0eb8713ef19eb6eedba17c3080a4165d5b933b",85    "floating_checkpoint_bytes": 7672043520,86    "floating_checkpoint_elements": 3836021760,87    "head_dim": 128,88    "hidden_size": 3072,89    "intermediate_size": 8192,90    "largest_transformer_layer_elements": 100669440,91    "largest_weight_tensor_elements": 614596608,92    "lora_rank": 64,93    "model_id": "microsoft/Phi-4-mini-instruct",94    "non_transformer_layer_elements": 614599680,95    "num_attention_heads": 24,96    "num_hidden_layers": 32,97    "num_key_value_heads": 8,98    "revision": "cfbefacb99257ffa30c83adab238a50856ac3083",99    "scope": "Numeric architecture/checkpoint-header inputs only; no weight payload. Adapter count assumes rank-64 all-linear transformer layers, excluding embeddings/output head. Floating checkpoint elements may include buffers; no separately allocated reference model is assumed.",100    "shard_header_sha256": {101      "model-00001-of-00002.safetensors": "29dd019fbd86197c6d12794e36555ab5bbd74a3b0acfe7c0736176bbea03a2ca",102      "model-00002-of-00002.safetensors": "54aed6db58d46324a40d42491b03f1e6d0c06da5022d61457d992c4f570d4c41"103    },104    "tensor_count": 194,105    "tie_word_embeddings": true,106    "vocab_size": 200064107  }108}109