Upload folder using huggingface_hub

Browse files

Files changed (5) hide show

config.json +2 -2
eval_results_2025-10-15T11-05-32.818241.json +185 -0
generation_config.json +1 -0
pytorch_model.bin +3 -0
training_args.bin +3 -0

config.json CHANGED Viewed

@@ -4,7 +4,8 @@
   ],
   "attention_bias": false,
   "attention_dropout": 0.0,
-  "dtype": "bfloat16",
   "eos_token_id": 151645,
   "head_dim": 128,
   "hidden_act": "silu",
@@ -55,7 +56,6 @@
   "num_attention_heads": 32,
   "num_hidden_layers": 36,
   "num_key_value_heads": 8,
-  "pad_token_id": 151643,
   "quantization_config": {
     "include_input_output_embeddings": true,
     "modules_to_not_convert": [

   ],
   "attention_bias": false,
   "attention_dropout": 0.0,
+  "bos_token_id": 151643,
+  "dtype": "float32",
   "eos_token_id": 151645,
   "head_dim": 128,
   "hidden_act": "silu",
   "num_attention_heads": 32,
   "num_hidden_layers": 36,
   "num_key_value_heads": 8,
   "quantization_config": {
     "include_input_output_embeddings": true,
     "modules_to_not_convert": [

eval_results_2025-10-15T11-05-32.818241.json ADDED Viewed

	@@ -0,0 +1,185 @@

+{
+  "results": {
+    "arc_challenge": {
+      "alias": "arc_challenge",
+      "acc,none": 0.39761092150170646,
+      "acc_stderr,none": 0.014301752223279512,
+      "acc_norm,none": 0.4197952218430034,
+      "acc_norm_stderr,none": 0.014422181226303012
+    },
+    "hellaswag": {
+      "alias": "hellaswag",
+      "acc,none": 0.45797649870543716,
+      "acc_stderr,none": 0.0049721265230316045,
+      "acc_norm,none": 0.6078470424218283,
+      "acc_norm_stderr,none": 0.004872326888655541
+    }
+  },
+  "group_subtasks": {
+    "arc_challenge": [],
+    "hellaswag": []
+  },
+  "configs": {
+    "arc_challenge": {
+      "task": "arc_challenge",
+      "tag": [
+        "ai2_arc"
+      ],
+      "dataset_path": "allenai/ai2_arc",
+      "dataset_name": "ARC-Challenge",
+      "training_split": "train",
+      "validation_split": "validation",
+      "test_split": "test",
+      "doc_to_text": "Question: {{question}}\nAnswer:",
+      "doc_to_target": "{{choices.label.index(answerKey)}}",
+      "unsafe_code": false,
+      "doc_to_choice": "{{choices.text}}",
+      "description": "",
+      "target_delimiter": " ",
+      "fewshot_delimiter": "\n\n",
+      "num_fewshot": 0,
+      "metric_list": [
+        {
+          "metric": "acc",
+          "aggregation": "mean",
+          "higher_is_better": true
+        },
+        {
+          "metric": "acc_norm",
+          "aggregation": "mean",
+          "higher_is_better": true
+        }
+      ],
+      "output_type": "multiple_choice",
+      "repeats": 1,
+      "should_decontaminate": true,
+      "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
+      "metadata": {
+        "version": 1.0,
+        "pretrained": "/home/lvj/local/checkpoints/qwen3-2bit-fineweb-26665/checkpoint-final/quant_converted",
+        "dtype": "auto",
+        "parallelize": false
+      }
+    },
+    "hellaswag": {
+      "task": "hellaswag",
+      "tag": [
+        "multiple_choice"
+      ],
+      "dataset_path": "Rowan/hellaswag",
+      "training_split": "train",
+      "validation_split": "validation",
+      "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n    def _process_doc(doc):\n        ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n        out_doc = {\n            \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n            \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n            \"gold\": int(doc[\"label\"]),\n        }\n        return out_doc\n\n    return dataset.map(_process_doc)\n",
+      "doc_to_text": "{{query}}",
+      "doc_to_target": "{{label}}",
+      "unsafe_code": false,
+      "doc_to_choice": "choices",
+      "description": "",
+      "target_delimiter": " ",
+      "fewshot_delimiter": "\n\n",
+      "num_fewshot": 0,
+      "metric_list": [
+        {
+          "metric": "acc",
+          "aggregation": "mean",
+          "higher_is_better": true
+        },
+        {
+          "metric": "acc_norm",
+          "aggregation": "mean",
+          "higher_is_better": true
+        }
+      ],
+      "output_type": "multiple_choice",
+      "repeats": 1,
+      "should_decontaminate": false,
+      "metadata": {
+        "version": 1.0,
+        "pretrained": "/home/lvj/local/checkpoints/qwen3-2bit-fineweb-26665/checkpoint-final/quant_converted",
+        "dtype": "auto",
+        "parallelize": false
+      }
+    }
+  },
+  "versions": {
+    "arc_challenge": 1.0,
+    "hellaswag": 1.0
+  },
+  "n-shot": {
+    "arc_challenge": 0,
+    "hellaswag": 0
+  },
+  "higher_is_better": {
+    "arc_challenge": {
+      "acc": true,
+      "acc_norm": true
+    },
+    "hellaswag": {
+      "acc": true,
+      "acc_norm": true
+    }
+  },
+  "n-samples": {
+    "hellaswag": {
+      "original": 10042,
+      "effective": 10042
+    },
+    "arc_challenge": {
+      "original": 1172,
+      "effective": 1172
+    }
+  },
+  "config": {
+    "model": "hf",
+    "model_args": "pretrained=/home/lvj/local/checkpoints/qwen3-2bit-fineweb-26665/checkpoint-final/quant_converted,dtype=auto,parallelize=False,trust_remote_code=True",
+    "model_num_parameters": 4022468096,
+    "model_dtype": "torch.float32",
+    "model_revision": "main",
+    "model_sha": "",
+    "batch_size": "auto",
+    "batch_sizes": [
+      64
+    ],
+    "device": null,
+    "use_cache": null,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "gen_kwargs": null,
+    "random_seed": 0,
+    "numpy_seed": 1234,
+    "torch_seed": 1234,
+    "fewshot_seed": 1234
+  },
+  "git_hash": "0726e21",
+  "date": 1760550653.300171,
+  "pretty_env_info": "PyTorch version: 2.7.0.dev20250310+cu124\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: CentOS Stream 9 (x86_64)\nGCC version: (GCC) 11.5.0 20240719 (Red Hat 11.5.0-11)\nClang version: Could not collect\nCMake version: Could not collect\nLibc version: glibc-2.34\n\nPython version: 3.13.5 | packaged by Anaconda, Inc. | (main, Jun 12 2025, 16:09:02) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-6.4.3-0_fbk15_hardened_2630_gf27365f948db-x86_64-with-glibc2.34\nIs CUDA available: True\nCUDA runtime version: 12.4.99\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA H100\nNvidia driver version: 550.90.07\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture:                       x86_64\nCPU op-mode(s):                     32-bit, 64-bit\nAddress sizes:                      52 bits physical, 57 bits virtual\nByte Order:                         Little Endian\nCPU(s):                             46\nOn-line CPU(s) list:                0-45\nVendor ID:                          AuthenticAMD\nModel name:                         AMD EPYC 9654 96-Core Processor\nCPU family:                         25\nModel:                              17\nThread(s) per core:                 1\nCore(s) per socket:                 46\nSocket(s):                          1\nStepping:                           1\nBogoMIPS:                           4792.80\nFlags:                              fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ht syscall nx mmxext fxsr_opt pdpe1gb rdtscp lm rep_good nopl cpuid extd_apicid tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm cmp_legacy svm cr8_legacy abm sse4a misalignsse 3dnowprefetch osvw perfctr_core invpcid_single ssbd ibrs ibpb stibp ibrs_enhanced vmmcall fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx512_bf16 clzero xsaveerptr wbnoinvd arat npt lbrv nrip_save tsc_scale vmcb_clean flushbyasid pausefilter pfthreshold v_vmsave_vmload vgif vnmi avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq rdpid fsrm flush_l1d arch_capabilities\nVirtualization:                     AMD-V\nHypervisor vendor:                  KVM\nVirtualization type:                full\nL1d cache:                          2.9 MiB (46 instances)\nL1i cache:                          2.9 MiB (46 instances)\nL2 cache:                           23 MiB (46 instances)\nL3 cache:                           736 MiB (46 instances)\nNUMA node(s):                       1\nNUMA node0 CPU(s):                  0-45\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit:        Not affected\nVulnerability L1tf:                 Not affected\nVulnerability Mds:                  Not affected\nVulnerability Meltdown:             Not affected\nVulnerability Mmio stale data:      Not affected\nVulnerability Retbleed:             Not affected\nVulnerability Spec store bypass:    Vulnerable\nVulnerability Spectre v1:           Vulnerable: __user pointer sanitization and usercopy barriers only; no swapgs barriers\nVulnerability Spectre v2:           Vulnerable, IBPB: disabled, STIBP: disabled, PBRSB-eIBRS: Not affected\nVulnerability Srbds:                Not affected\nVulnerability Tsx async abort:      Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.3.3\n[pip3] nvidia-cublas-cu12==12.4.5.8\n[pip3] nvidia-cuda-cupti-cu12==12.4.127\n[pip3] nvidia-cuda-nvrtc-cu12==12.4.127\n[pip3] nvidia-cuda-runtime-cu12==12.4.127\n[pip3] nvidia-cudnn-cu12==9.1.0.70\n[pip3] nvidia-cufft-cu12==11.2.1.3\n[pip3] nvidia-curand-cu12==10.3.5.147\n[pip3] nvidia-cusolver-cu12==11.6.1.9\n[pip3] nvidia-cusparse-cu12==12.3.1.170\n[pip3] nvidia-cusparselt-cu12==0.6.2\n[pip3] nvidia-nccl-cu12==2.25.1\n[pip3] nvidia-nvjitlink-cu12==12.4.127\n[pip3] nvidia-nvtx-cu12==12.4.127\n[pip3] pytorch-triton==3.2.0+git4b3bb1f8\n[pip3] torch==2.7.0.dev20250310+cu124\n[pip3] torchao==0.14.0+gitabc60fde\n[pip3] triton==3.4.0\n[conda] No relevant packages",
+  "transformers_version": "4.56.2",
+  "lm_eval_version": "0.4.9.1",
+  "upper_git_hash": null,
+  "tokenizer_pad_token": [
+    "<|endoftext|>",
+    "151643"
+  ],
+  "tokenizer_eos_token": [
+    "<|im_end|>",
+    "151645"
+  ],
+  "tokenizer_bos_token": [
+    null,
+    "None"
+  ],
+  "eot_token_id": 151645,
+  "max_length": 40960,
+  "task_hashes": {},
+  "model_source": "hf",
+  "model_name": "/home/lvj/local/checkpoints/qwen3-2bit-fineweb-26665/checkpoint-final/quant_converted",
+  "model_name_sanitized": "__home__lvj__local__checkpoints__qwen3-2bit-fineweb-26665__checkpoint-final__quant_converted",
+  "system_instruction": null,
+  "system_instruction_sha": null,
+  "fewshot_as_multiturn": false,
+  "chat_template": null,
+  "chat_template_sha": null,
+  "start_time": 1197762.297058725,
+  "end_time": 1198643.783406351,
+  "total_evaluation_time_seconds": "881.4863476259634"
+}

generation_config.json CHANGED Viewed

@@ -1,4 +1,5 @@
 {
   "do_sample": true,
   "eos_token_id": [
     151645,

 {
+  "bos_token_id": 151643,
   "do_sample": true,
   "eos_token_id": [
     151645,

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3440bc2a0ed0b49b84156c630012aa4de46ebc28c58b778b5a4826b838720ea0
+size 4422692167

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bf7e8bd9cb3edb141046c29e62667a8ee60965fb14a23dbc504abda991e29c85
+size 6417