PyTorch
English
llama
richardmfan commited on
Commit
374e054
·
verified ·
1 Parent(s): c597b86

Upload mid_1_0065000

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +5 -0
  2. config.json +3 -2
  3. done.txt +1 -1
  4. eval_results/arc_challenge_25shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-05-55.784564.json +125 -0
  5. eval_results/arc_challenge_25shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_arc_challenge_2025-08-05T08-05-55.784564.jsonl +3 -0
  6. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-26-11.797689.json +0 -0
  7. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_boolean_expressions_2025-08-05T08-26-11.797689.jsonl +0 -0
  8. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_causal_judgement_2025-08-05T08-26-11.797689.jsonl +0 -0
  9. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_date_understanding_2025-08-05T08-26-11.797689.jsonl +0 -0
  10. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_disambiguation_qa_2025-08-05T08-26-11.797689.jsonl +0 -0
  11. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_dyck_languages_2025-08-05T08-26-11.797689.jsonl +0 -0
  12. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_formal_fallacies_2025-08-05T08-26-11.797689.jsonl +0 -0
  13. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_geometric_shapes_2025-08-05T08-26-11.797689.jsonl +0 -0
  14. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_hyperbaton_2025-08-05T08-26-11.797689.jsonl +0 -0
  15. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_logical_deduction_five_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  16. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_logical_deduction_seven_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  17. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_logical_deduction_three_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  18. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_movie_recommendation_2025-08-05T08-26-11.797689.jsonl +0 -0
  19. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_multistep_arithmetic_two_2025-08-05T08-26-11.797689.jsonl +0 -0
  20. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_navigate_2025-08-05T08-26-11.797689.jsonl +0 -0
  21. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_object_counting_2025-08-05T08-26-11.797689.jsonl +0 -0
  22. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_penguins_in_a_table_2025-08-05T08-26-11.797689.jsonl +0 -0
  23. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_reasoning_about_colored_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  24. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_ruin_names_2025-08-05T08-26-11.797689.jsonl +0 -0
  25. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_salient_translation_error_detection_2025-08-05T08-26-11.797689.jsonl +0 -0
  26. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_snarks_2025-08-05T08-26-11.797689.jsonl +0 -0
  27. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_sports_understanding_2025-08-05T08-26-11.797689.jsonl +0 -0
  28. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_temporal_sequences_2025-08-05T08-26-11.797689.jsonl +0 -0
  29. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_tracking_shuffled_objects_five_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  30. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_tracking_shuffled_objects_seven_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  31. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_tracking_shuffled_objects_three_objects_2025-08-05T08-26-11.797689.jsonl +0 -0
  32. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_web_of_lies_2025-08-05T08-26-11.797689.jsonl +0 -0
  33. eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_word_sorting_2025-08-05T08-26-11.797689.jsonl +0 -0
  34. eval_results/gsm8k_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-57-10.856012.json +161 -0
  35. eval_results/gsm8k_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_gsm8k_2025-08-05T08-57-10.856012.jsonl +3 -0
  36. eval_results/gsm8k_cot_8shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-06T21-19-21.441609.json +195 -0
  37. eval_results/gsm8k_cot_8shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_gsm8k_cot_2025-08-06T21-19-21.441609.jsonl +0 -0
  38. eval_results/hellaswag_10shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T12-06-56.401273.json +126 -0
  39. eval_results/hellaswag_10shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_hellaswag_2025-08-05T12-06-56.401273.jsonl +3 -0
  40. eval_results/humaneval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T07-34-12.673211.json +134 -0
  41. eval_results/humaneval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_humaneval_2025-08-05T07-34-12.673211.jsonl +0 -0
  42. eval_results/ifeval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-21-28.685979.json +140 -0
  43. eval_results/ifeval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_ifeval_2025-08-05T08-21-28.685979.jsonl +0 -0
  44. eval_results/leaderboard_gpqa_diamond/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-31-47.504073.json +121 -0
  45. eval_results/leaderboard_gpqa_diamond/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_leaderboard_gpqa_diamond_2025-08-05T08-31-47.504073.jsonl +0 -0
  46. eval_results/mbpp_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T07-47-32.772547.json +120 -0
  47. eval_results/mbpp_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_mbpp_2025-08-05T07-47-32.772547.jsonl +0 -0
  48. eval_results/minerva_math_4shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-14T03-07-36.370756.json +584 -0
  49. eval_results/minerva_math_4shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_minerva_math_algebra_2025-08-14T03-07-36.370756.jsonl +0 -0
  50. eval_results/minerva_math_4shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_minerva_math_counting_and_prob_2025-08-14T03-07-36.370756.jsonl +0 -0
.gitattributes CHANGED
@@ -51,3 +51,8 @@ eval_results/gsm8k_5shots/__lustrefs__users__runner__checkpoints__huggingface__i
51
  eval_results/hellaswag_10shots/__lustrefs__users__runner__checkpoints__huggingface__iter_0020000/samples_hellaswag_2025-04-03T10-46-52.083797.jsonl filter=lfs diff=lfs merge=lfs -text
52
  eval_results/mmlu_5shots/__lustrefs__users__runner__checkpoints__huggingface__iter_0020000/samples_mmlu_professional_law_2025-04-03T11-38-41.517466.jsonl filter=lfs diff=lfs merge=lfs -text
53
  eval_results/mmlu_pro_5shots/__lustrefs__users__runner__checkpoints__huggingface__iter_0020000/samples_mmlu_pro_law_2025-04-09T05-02-59.883377.jsonl filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
51
  eval_results/hellaswag_10shots/__lustrefs__users__runner__checkpoints__huggingface__iter_0020000/samples_hellaswag_2025-04-03T10-46-52.083797.jsonl filter=lfs diff=lfs merge=lfs -text
52
  eval_results/mmlu_5shots/__lustrefs__users__runner__checkpoints__huggingface__iter_0020000/samples_mmlu_professional_law_2025-04-03T11-38-41.517466.jsonl filter=lfs diff=lfs merge=lfs -text
53
  eval_results/mmlu_pro_5shots/__lustrefs__users__runner__checkpoints__huggingface__iter_0020000/samples_mmlu_pro_law_2025-04-09T05-02-59.883377.jsonl filter=lfs diff=lfs merge=lfs -text
54
+ eval_results/arc_challenge_25shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_arc_challenge_2025-08-05T08-05-55.784564.jsonl filter=lfs diff=lfs merge=lfs -text
55
+ eval_results/gsm8k_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_gsm8k_2025-08-05T08-57-10.856012.jsonl filter=lfs diff=lfs merge=lfs -text
56
+ eval_results/hellaswag_10shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_hellaswag_2025-08-05T12-06-56.401273.jsonl filter=lfs diff=lfs merge=lfs -text
57
+ eval_results/mmlu_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_mmlu_professional_law_2025-08-05T12-43-23.697649.jsonl filter=lfs diff=lfs merge=lfs -text
58
+ eval_results/mmlu_pro_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_mmlu_pro_law_2025-08-05T09-47-13.715353.jsonl filter=lfs diff=lfs merge=lfs -text
config.json CHANGED
@@ -23,7 +23,8 @@
23
  "rope_scaling": null,
24
  "rope_theta": 500000,
25
  "tie_word_embeddings": false,
26
- "transformers_version": "4.49.0",
 
27
  "use_cache": true,
28
- "vocab_size": 250880
29
  }
 
23
  "rope_scaling": null,
24
  "rope_theta": 500000,
25
  "tie_word_embeddings": false,
26
+ "torch_dtype": "float32",
27
+ "transformers_version": "4.53.2",
28
  "use_cache": true,
29
+ "vocab_size": 250112
30
  }
done.txt CHANGED
@@ -1 +1 @@
1
- 2025-07-08 21:39:09.403951-05:00
 
1
+ saving done
eval_results/arc_challenge_25shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-05-55.784564.json ADDED
@@ -0,0 +1,125 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.6296928327645052,
6
+ "acc_stderr,none": 0.014111298751675015,
7
+ "acc_norm,none": 0.6629692832764505,
8
+ "acc_norm_stderr,none": 0.013813476652902168
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_challenge": []
13
+ },
14
+ "configs": {
15
+ "arc_challenge": {
16
+ "task": "arc_challenge",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Challenge",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 25,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
52
+ "tensor_parallel_size": 8,
53
+ "dtype": "float32",
54
+ "gpu_memory_utilization": 0.8
55
+ }
56
+ }
57
+ },
58
+ "versions": {
59
+ "arc_challenge": 1.0
60
+ },
61
+ "n-shot": {
62
+ "arc_challenge": 25
63
+ },
64
+ "higher_is_better": {
65
+ "arc_challenge": {
66
+ "acc": true,
67
+ "acc_norm": true
68
+ }
69
+ },
70
+ "n-samples": {
71
+ "arc_challenge": {
72
+ "original": 1172,
73
+ "effective": 1172
74
+ }
75
+ },
76
+ "config": {
77
+ "model": "vllm",
78
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
79
+ "batch_size": "1",
80
+ "batch_sizes": [],
81
+ "device": null,
82
+ "use_cache": null,
83
+ "limit": null,
84
+ "bootstrap_iters": 100000,
85
+ "gen_kwargs": null,
86
+ "random_seed": 0,
87
+ "numpy_seed": 1234,
88
+ "torch_seed": 1234,
89
+ "fewshot_seed": 1234
90
+ },
91
+ "git_hash": "18965e2",
92
+ "date": 1754378592.6596196,
93
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 4000.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
94
+ "transformers_version": "4.53.0.dev0",
95
+ "lm_eval_version": "0.4.8",
96
+ "upper_git_hash": null,
97
+ "tokenizer_pad_token": [
98
+ "<|end_of_text|>",
99
+ "1"
100
+ ],
101
+ "tokenizer_eos_token": [
102
+ "<|end_of_text|>",
103
+ "1"
104
+ ],
105
+ "tokenizer_bos_token": [
106
+ "<|begin_of_text|>",
107
+ "0"
108
+ ],
109
+ "eot_token_id": 1,
110
+ "max_length": 32768,
111
+ "task_hashes": {
112
+ "arc_challenge": "55e883475b5650b20d8d9dc1e9cdf59ef645a257fcd74bf43a9dbb5c632c529c"
113
+ },
114
+ "model_source": "vllm",
115
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
116
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
117
+ "system_instruction": null,
118
+ "system_instruction_sha": null,
119
+ "fewshot_as_multiturn": false,
120
+ "chat_template": null,
121
+ "chat_template_sha": null,
122
+ "start_time": 3069830.225650459,
123
+ "end_time": 3072412.07092101,
124
+ "total_evaluation_time_seconds": "2581.84527055081"
125
+ }
eval_results/arc_challenge_25shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_arc_challenge_2025-08-05T08-05-55.784564.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9399be445cfd33cbcf5f77afd6b24a873fe2536fe68e51d69c2f2cc2b2df74df
3
+ size 23434070
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-26-11.797689.json ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_boolean_expressions_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_causal_judgement_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_date_understanding_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_disambiguation_qa_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_dyck_languages_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_formal_fallacies_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_geometric_shapes_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_hyperbaton_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_logical_deduction_five_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_logical_deduction_seven_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_logical_deduction_three_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_movie_recommendation_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_multistep_arithmetic_two_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_navigate_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_object_counting_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_penguins_in_a_table_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_reasoning_about_colored_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_ruin_names_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_salient_translation_error_detection_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_snarks_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_sports_understanding_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_temporal_sequences_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_tracking_shuffled_objects_five_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_tracking_shuffled_objects_seven_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_tracking_shuffled_objects_three_objects_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_web_of_lies_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/bbh_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_bbh_cot_fewshot_word_sorting_2025-08-05T08-26-11.797689.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/gsm8k_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-57-10.856012.json ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k": {
4
+ "alias": "gsm8k",
5
+ "exact_match,strict-match": 0.6504927975739196,
6
+ "exact_match_stderr,strict-match": 0.013133836511705995,
7
+ "exact_match,flexible-extract": 0.66868840030326,
8
+ "exact_match_stderr,flexible-extract": 0.012964999679688664
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "gsm8k": []
13
+ },
14
+ "configs": {
15
+ "gsm8k": {
16
+ "task": "gsm8k",
17
+ "tag": [
18
+ "math_word_problems"
19
+ ],
20
+ "dataset_path": "gsm8k",
21
+ "dataset_name": "main",
22
+ "training_split": "train",
23
+ "test_split": "test",
24
+ "fewshot_split": "train",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{answer}}",
27
+ "unsafe_code": false,
28
+ "description": "",
29
+ "target_delimiter": " ",
30
+ "fewshot_delimiter": "\n\n",
31
+ "num_fewshot": 5,
32
+ "metric_list": [
33
+ {
34
+ "metric": "exact_match",
35
+ "aggregation": "mean",
36
+ "higher_is_better": true,
37
+ "ignore_case": true,
38
+ "ignore_punctuation": false,
39
+ "regexes_to_ignore": [
40
+ ",",
41
+ "\\$",
42
+ "(?s).*#### ",
43
+ "\\.$"
44
+ ]
45
+ }
46
+ ],
47
+ "output_type": "generate_until",
48
+ "generation_kwargs": {
49
+ "until": [
50
+ "Question:",
51
+ "</s>",
52
+ "<|im_end|>"
53
+ ],
54
+ "do_sample": false,
55
+ "temperature": 0.0
56
+ },
57
+ "repeats": 1,
58
+ "filter_list": [
59
+ {
60
+ "name": "strict-match",
61
+ "filter": [
62
+ {
63
+ "function": "regex",
64
+ "regex_pattern": "#### (\\-?[0-9\\.\\,]+)"
65
+ },
66
+ {
67
+ "function": "take_first"
68
+ }
69
+ ]
70
+ },
71
+ {
72
+ "name": "flexible-extract",
73
+ "filter": [
74
+ {
75
+ "function": "regex",
76
+ "group_select": -1,
77
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
78
+ },
79
+ {
80
+ "function": "take_first"
81
+ }
82
+ ]
83
+ }
84
+ ],
85
+ "should_decontaminate": false,
86
+ "metadata": {
87
+ "version": 3.0,
88
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
89
+ "tensor_parallel_size": 8,
90
+ "dtype": "float32",
91
+ "gpu_memory_utilization": 0.8
92
+ }
93
+ }
94
+ },
95
+ "versions": {
96
+ "gsm8k": 3.0
97
+ },
98
+ "n-shot": {
99
+ "gsm8k": 5
100
+ },
101
+ "higher_is_better": {
102
+ "gsm8k": {
103
+ "exact_match": true
104
+ }
105
+ },
106
+ "n-samples": {
107
+ "gsm8k": {
108
+ "original": 1319,
109
+ "effective": 1319
110
+ }
111
+ },
112
+ "config": {
113
+ "model": "vllm",
114
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
115
+ "batch_size": "1",
116
+ "batch_sizes": [],
117
+ "device": null,
118
+ "use_cache": null,
119
+ "limit": null,
120
+ "bootstrap_iters": 100000,
121
+ "gen_kwargs": null,
122
+ "random_seed": 0,
123
+ "numpy_seed": 1234,
124
+ "torch_seed": 1234,
125
+ "fewshot_seed": 1234
126
+ },
127
+ "git_hash": "18965e2",
128
+ "date": 1754381182.2477229,
129
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 4000.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
130
+ "transformers_version": "4.53.0.dev0",
131
+ "lm_eval_version": "0.4.8",
132
+ "upper_git_hash": null,
133
+ "tokenizer_pad_token": [
134
+ "<|end_of_text|>",
135
+ "1"
136
+ ],
137
+ "tokenizer_eos_token": [
138
+ "<|end_of_text|>",
139
+ "1"
140
+ ],
141
+ "tokenizer_bos_token": [
142
+ "<|begin_of_text|>",
143
+ "0"
144
+ ],
145
+ "eot_token_id": 1,
146
+ "max_length": 32768,
147
+ "task_hashes": {
148
+ "gsm8k": "2330f4ebfcccaf66a892922df2819cdb1f118e448d076d3f42bdde4177678ac7"
149
+ },
150
+ "model_source": "vllm",
151
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
152
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
153
+ "system_instruction": null,
154
+ "system_instruction_sha": null,
155
+ "fewshot_as_multiturn": false,
156
+ "chat_template": null,
157
+ "chat_template_sha": null,
158
+ "start_time": 3072433.087531156,
159
+ "end_time": 3075487.145490341,
160
+ "total_evaluation_time_seconds": "3054.057959184982"
161
+ }
eval_results/gsm8k_5shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_gsm8k_2025-08-05T08-57-10.856012.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:18069e12f34a44ee464407784cf17aeccfcb7076633ec73996e3d1e3377c3263
3
+ size 12552189
eval_results/gsm8k_cot_8shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-06T21-19-21.441609.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "gsm8k_cot": {
4
+ "alias": "gsm8k_cot",
5
+ "exact_match,strict-match": 0.7270659590598939,
6
+ "exact_match_stderr,strict-match": 0.012270381151108758,
7
+ "exact_match,flexible-extract": 0.7338893100833965,
8
+ "exact_match_stderr,flexible-extract": 0.012172750939040316
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "gsm8k_cot": []
13
+ },
14
+ "configs": {
15
+ "gsm8k_cot": {
16
+ "task": "gsm8k_cot",
17
+ "tag": [
18
+ "chain_of_thought"
19
+ ],
20
+ "dataset_path": "gsm8k",
21
+ "dataset_name": "main",
22
+ "test_split": "test",
23
+ "doc_to_text": "Q: {{question}}\nA:",
24
+ "doc_to_target": "{{answer.split('####')[-1].strip() if answer is defined else target}}",
25
+ "unsafe_code": false,
26
+ "description": "",
27
+ "target_delimiter": " ",
28
+ "fewshot_delimiter": "\n\n",
29
+ "fewshot_config": {
30
+ "sampler": "first_n",
31
+ "samples": [
32
+ {
33
+ "question": "There are 15 trees in the grove. Grove workers will plant trees in the grove today. After they are done, there will be 21 trees. How many trees did the grove workers plant today?",
34
+ "target": "There are 15 trees originally. Then there were 21 trees after some more were planted. So there must have been 21 - 15 = 6. The answer is 6."
35
+ },
36
+ {
37
+ "question": "If there are 3 cars in the parking lot and 2 more cars arrive, how many cars are in the parking lot?",
38
+ "target": "There are originally 3 cars. 2 more cars arrive. 3 + 2 = 5. The answer is 5."
39
+ },
40
+ {
41
+ "question": "Leah had 32 chocolates and her sister had 42. If they ate 35, how many pieces do they have left in total?",
42
+ "target": "Originally, Leah had 32 chocolates. Her sister had 42. So in total they had 32 + 42 = 74. After eating 35, they had 74 - 35 = 39. The answer is 39."
43
+ },
44
+ {
45
+ "question": "Jason had 20 lollipops. He gave Denny some lollipops. Now Jason has 12 lollipops. How many lollipops did Jason give to Denny?",
46
+ "target": "Jason started with 20 lollipops. Then he had 12 after giving some to Denny. So he gave Denny 20 - 12 = 8. The answer is 8."
47
+ },
48
+ {
49
+ "question": "Shawn has five toys. For Christmas, he got two toys each from his mom and dad. How many toys does he have now?",
50
+ "target": "Shawn started with 5 toys. If he got 2 toys each from his mom and dad, then that is 4 more toys. 5 + 4 = 9. The answer is 9."
51
+ },
52
+ {
53
+ "question": "There were nine computers in the server room. Five more computers were installed each day, from monday to thursday. How many computers are now in the server room?",
54
+ "target": "There were originally 9 computers. For each of 4 days, 5 more computers were added. So 5 * 4 = 20 computers were added. 9 + 20 is 29. The answer is 29."
55
+ },
56
+ {
57
+ "question": "Michael had 58 golf balls. On tuesday, he lost 23 golf balls. On wednesday, he lost 2 more. How many golf balls did he have at the end of wednesday?",
58
+ "target": "Michael started with 58 golf balls. After losing 23 on tuesday, he had 58 - 23 = 35. After losing 2 more, he had 35 - 2 = 33 golf balls. The answer is 33."
59
+ },
60
+ {
61
+ "question": "Olivia has $23. She bought five bagels for $3 each. How much money does she have left?",
62
+ "target": "Olivia had 23 dollars. 5 bagels for 3 dollars each will be 5 x 3 = 15 dollars. So she has 23 - 15 dollars left. 23 - 15 is 8. The answer is 8."
63
+ }
64
+ ]
65
+ },
66
+ "num_fewshot": 8,
67
+ "metric_list": [
68
+ {
69
+ "aggregation": "mean",
70
+ "higher_is_better": true,
71
+ "ignore_case": true,
72
+ "ignore_punctuation": false,
73
+ "metric": "exact_match",
74
+ "regexes_to_ignore": [
75
+ ",",
76
+ "\\$",
77
+ "(?s).*#### ",
78
+ "\\.$"
79
+ ]
80
+ }
81
+ ],
82
+ "output_type": "generate_until",
83
+ "generation_kwargs": {
84
+ "do_sample": false,
85
+ "until": [
86
+ "Q:",
87
+ "</s>",
88
+ "<|im_end|>"
89
+ ]
90
+ },
91
+ "repeats": 1,
92
+ "filter_list": [
93
+ {
94
+ "filter": [
95
+ {
96
+ "function": "regex",
97
+ "regex_pattern": "The answer is (\\-?[0-9\\.\\,]+)."
98
+ },
99
+ {
100
+ "function": "take_first"
101
+ }
102
+ ],
103
+ "name": "strict-match"
104
+ },
105
+ {
106
+ "filter": [
107
+ {
108
+ "function": "regex",
109
+ "group_select": -1,
110
+ "regex_pattern": "(-?[$0-9.,]{2,})|(-?[0-9]+)"
111
+ },
112
+ {
113
+ "function": "take_first"
114
+ }
115
+ ],
116
+ "name": "flexible-extract"
117
+ }
118
+ ],
119
+ "should_decontaminate": false,
120
+ "metadata": {
121
+ "version": 3.0,
122
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
123
+ "tensor_parallel_size": 8,
124
+ "dtype": "float32",
125
+ "gpu_memory_utilization": 0.8
126
+ }
127
+ }
128
+ },
129
+ "versions": {
130
+ "gsm8k_cot": 3.0
131
+ },
132
+ "n-shot": {
133
+ "gsm8k_cot": 8
134
+ },
135
+ "higher_is_better": {
136
+ "gsm8k_cot": {
137
+ "exact_match": true
138
+ }
139
+ },
140
+ "n-samples": {
141
+ "gsm8k_cot": {
142
+ "original": 1319,
143
+ "effective": 1319
144
+ }
145
+ },
146
+ "config": {
147
+ "model": "vllm",
148
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
149
+ "batch_size": "auto",
150
+ "batch_sizes": [],
151
+ "device": null,
152
+ "use_cache": null,
153
+ "limit": null,
154
+ "bootstrap_iters": 100000,
155
+ "gen_kwargs": null,
156
+ "random_seed": 0,
157
+ "numpy_seed": 1234,
158
+ "torch_seed": 1234,
159
+ "fewshot_seed": 1234
160
+ },
161
+ "git_hash": "18965e2",
162
+ "date": 1754514366.5816486,
163
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 3999.99\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
164
+ "transformers_version": "4.53.0.dev0",
165
+ "lm_eval_version": "0.4.8",
166
+ "upper_git_hash": null,
167
+ "tokenizer_pad_token": [
168
+ "<|end_of_text|>",
169
+ "1"
170
+ ],
171
+ "tokenizer_eos_token": [
172
+ "<|end_of_text|>",
173
+ "1"
174
+ ],
175
+ "tokenizer_bos_token": [
176
+ "<|begin_of_text|>",
177
+ "0"
178
+ ],
179
+ "eot_token_id": 1,
180
+ "max_length": 32768,
181
+ "task_hashes": {
182
+ "gsm8k_cot": "fc360963b39ee52c26a82795124f9ad7da4d6a8fecf1b77e2502823b1669b3d0"
183
+ },
184
+ "model_source": "vllm",
185
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
186
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
187
+ "system_instruction": null,
188
+ "system_instruction_sha": null,
189
+ "fewshot_as_multiturn": false,
190
+ "chat_template": null,
191
+ "chat_template_sha": null,
192
+ "start_time": 114623.472364141,
193
+ "end_time": 115437.804433137,
194
+ "total_evaluation_time_seconds": "814.3320689960092"
195
+ }
eval_results/gsm8k_cot_8shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_gsm8k_cot_2025-08-06T21-19-21.441609.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/hellaswag_10shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T12-06-56.401273.json ADDED
@@ -0,0 +1,126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "hellaswag": {
4
+ "alias": "hellaswag",
5
+ "acc,none": 0.6805417247560247,
6
+ "acc_stderr,none": 0.00465313836094811,
7
+ "acc_norm,none": 0.8678550089623581,
8
+ "acc_norm_stderr,none": 0.0033795622983875803
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "hellaswag": []
13
+ },
14
+ "configs": {
15
+ "hellaswag": {
16
+ "task": "hellaswag",
17
+ "tag": [
18
+ "multiple_choice"
19
+ ],
20
+ "dataset_path": "hellaswag",
21
+ "dataset_kwargs": {
22
+ "trust_remote_code": true
23
+ },
24
+ "training_split": "train",
25
+ "validation_split": "validation",
26
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
27
+ "doc_to_text": "{{query}}",
28
+ "doc_to_target": "{{label}}",
29
+ "unsafe_code": false,
30
+ "doc_to_choice": "choices",
31
+ "description": "",
32
+ "target_delimiter": " ",
33
+ "fewshot_delimiter": "\n\n",
34
+ "num_fewshot": 10,
35
+ "metric_list": [
36
+ {
37
+ "metric": "acc",
38
+ "aggregation": "mean",
39
+ "higher_is_better": true
40
+ },
41
+ {
42
+ "metric": "acc_norm",
43
+ "aggregation": "mean",
44
+ "higher_is_better": true
45
+ }
46
+ ],
47
+ "output_type": "multiple_choice",
48
+ "repeats": 1,
49
+ "should_decontaminate": false,
50
+ "metadata": {
51
+ "version": 1.0,
52
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
53
+ "tensor_parallel_size": 8,
54
+ "dtype": "float32",
55
+ "gpu_memory_utilization": 0.8
56
+ }
57
+ }
58
+ },
59
+ "versions": {
60
+ "hellaswag": 1.0
61
+ },
62
+ "n-shot": {
63
+ "hellaswag": 10
64
+ },
65
+ "higher_is_better": {
66
+ "hellaswag": {
67
+ "acc": true,
68
+ "acc_norm": true
69
+ }
70
+ },
71
+ "n-samples": {
72
+ "hellaswag": {
73
+ "original": 10042,
74
+ "effective": 10042
75
+ }
76
+ },
77
+ "config": {
78
+ "model": "vllm",
79
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
80
+ "batch_size": "1",
81
+ "batch_sizes": [],
82
+ "device": null,
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "18965e2",
93
+ "date": 1754378629.4031453,
94
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 4000.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
95
+ "transformers_version": "4.53.0.dev0",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<|end_of_text|>",
100
+ "1"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "<|end_of_text|>",
104
+ "1"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<|begin_of_text|>",
108
+ "0"
109
+ ],
110
+ "eot_token_id": 1,
111
+ "max_length": 32768,
112
+ "task_hashes": {
113
+ "hellaswag": "d4bcb44ec68db2b8a65f050c3c64c48454179b48fd8aee3e73b55e2ec51e6d82"
114
+ },
115
+ "model_source": "vllm",
116
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
117
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
118
+ "system_instruction": null,
119
+ "system_instruction_sha": null,
120
+ "fewshot_as_multiturn": false,
121
+ "chat_template": null,
122
+ "chat_template_sha": null,
123
+ "start_time": 3069856.048769722,
124
+ "end_time": 3086862.056837167,
125
+ "total_evaluation_time_seconds": "17006.0080674449"
126
+ }
eval_results/hellaswag_10shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_hellaswag_2025-08-05T12-06-56.401273.jsonl ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0f317748b7a3856dfa8f813c1b07f2fc6f709ecc91065dd646ba29a51e9e7ded
3
+ size 187317064
eval_results/humaneval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T07-34-12.673211.json ADDED
@@ -0,0 +1,134 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "humaneval": {
4
+ "alias": "humaneval",
5
+ "pass@1,create_test": 0.5365853658536586,
6
+ "pass@1_stderr,create_test": 0.03905804324801912
7
+ }
8
+ },
9
+ "group_subtasks": {
10
+ "humaneval": []
11
+ },
12
+ "configs": {
13
+ "humaneval": {
14
+ "task": "humaneval",
15
+ "dataset_path": "openai/openai_humaneval",
16
+ "test_split": "test",
17
+ "doc_to_text": "{{prompt}}",
18
+ "doc_to_target": "{{test}}\ncheck({{entry_point}})",
19
+ "unsafe_code": true,
20
+ "description": "",
21
+ "target_delimiter": " ",
22
+ "fewshot_delimiter": "\n\n",
23
+ "num_fewshot": 0,
24
+ "metric_list": [
25
+ {
26
+ "metric": "def pass_at_k(references: list[str], predictions: list[list[str]], k: list[int] = None):\n global compute_\n assert k is not None\n if isinstance(k, int):\n k = [k]\n res = compute_.compute(\n references=references,\n predictions=predictions,\n k=k,\n )\n return res[0]\n",
27
+ "aggregation": "mean",
28
+ "higher_is_better": true,
29
+ "k": [
30
+ 1
31
+ ]
32
+ }
33
+ ],
34
+ "output_type": "generate_until",
35
+ "generation_kwargs": {
36
+ "until": [
37
+ "\nclass",
38
+ "\ndef",
39
+ "\n#",
40
+ "\nif",
41
+ "\nprint"
42
+ ],
43
+ "max_gen_toks": 1024,
44
+ "do_sample": false
45
+ },
46
+ "repeats": 1,
47
+ "filter_list": [
48
+ {
49
+ "name": "create_test",
50
+ "filter": [
51
+ {
52
+ "function": "custom",
53
+ "filter_fn": "<function build_predictions at 0x1537c5ac7e20>"
54
+ }
55
+ ]
56
+ }
57
+ ],
58
+ "should_decontaminate": false,
59
+ "metadata": {
60
+ "version": 1.0,
61
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
62
+ "tensor_parallel_size": 8,
63
+ "dtype": "float32",
64
+ "gpu_memory_utilization": 0.8
65
+ }
66
+ }
67
+ },
68
+ "versions": {
69
+ "humaneval": 1.0
70
+ },
71
+ "n-shot": {
72
+ "humaneval": 0
73
+ },
74
+ "higher_is_better": {
75
+ "humaneval": {
76
+ "pass_at_k": true
77
+ }
78
+ },
79
+ "n-samples": {
80
+ "humaneval": {
81
+ "original": 164,
82
+ "effective": 164
83
+ }
84
+ },
85
+ "config": {
86
+ "model": "vllm",
87
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
88
+ "batch_size": "1",
89
+ "batch_sizes": [],
90
+ "device": null,
91
+ "use_cache": null,
92
+ "limit": null,
93
+ "bootstrap_iters": 100000,
94
+ "gen_kwargs": null,
95
+ "random_seed": 0,
96
+ "numpy_seed": 1234,
97
+ "torch_seed": 1234,
98
+ "fewshot_seed": 1234
99
+ },
100
+ "git_hash": "18965e2",
101
+ "date": 1754378623.0225909,
102
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 4000.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
103
+ "transformers_version": "4.53.0.dev0",
104
+ "lm_eval_version": "0.4.8",
105
+ "upper_git_hash": null,
106
+ "tokenizer_pad_token": [
107
+ "<|end_of_text|>",
108
+ "1"
109
+ ],
110
+ "tokenizer_eos_token": [
111
+ "<|end_of_text|>",
112
+ "1"
113
+ ],
114
+ "tokenizer_bos_token": [
115
+ "<|begin_of_text|>",
116
+ "0"
117
+ ],
118
+ "eot_token_id": 1,
119
+ "max_length": 32768,
120
+ "task_hashes": {
121
+ "humaneval": "c122632f0bdffdd0162cc8033a879dfe9d3b35e82c52eda6a767069e0a2beb40"
122
+ },
123
+ "model_source": "vllm",
124
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
125
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
126
+ "system_instruction": null,
127
+ "system_instruction_sha": null,
128
+ "fewshot_as_multiturn": false,
129
+ "chat_template": null,
130
+ "chat_template_sha": null,
131
+ "start_time": 3069852.451609627,
132
+ "end_time": 3070499.901276076,
133
+ "total_evaluation_time_seconds": "647.4496664493345"
134
+ }
eval_results/humaneval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_humaneval_2025-08-05T07-34-12.673211.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/ifeval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-21-28.685979.json ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "ifeval": {
4
+ "alias": "ifeval",
5
+ "prompt_level_strict_acc,none": 0.18114602587800369,
6
+ "prompt_level_strict_acc_stderr,none": 0.01657374894371447,
7
+ "inst_level_strict_acc,none": 0.3129496402877698,
8
+ "inst_level_strict_acc_stderr,none": "N/A",
9
+ "prompt_level_loose_acc,none": 0.2033271719038817,
10
+ "prompt_level_loose_acc_stderr,none": 0.017319718641834708,
11
+ "inst_level_loose_acc,none": 0.32973621103117506,
12
+ "inst_level_loose_acc_stderr,none": "N/A"
13
+ }
14
+ },
15
+ "group_subtasks": {
16
+ "ifeval": []
17
+ },
18
+ "configs": {
19
+ "ifeval": {
20
+ "task": "ifeval",
21
+ "dataset_path": "google/IFEval",
22
+ "test_split": "train",
23
+ "doc_to_text": "prompt",
24
+ "doc_to_target": 0,
25
+ "unsafe_code": false,
26
+ "process_results": "def process_results(doc, results):\n inp = InputExample(\n key=doc[\"key\"],\n instruction_id_list=doc[\"instruction_id_list\"],\n prompt=doc[\"prompt\"],\n kwargs=doc[\"kwargs\"],\n )\n response = results[0]\n\n out_strict = test_instruction_following_strict(inp, response)\n out_loose = test_instruction_following_loose(inp, response)\n\n return {\n \"prompt_level_strict_acc\": out_strict.follow_all_instructions,\n \"inst_level_strict_acc\": out_strict.follow_instruction_list,\n \"prompt_level_loose_acc\": out_loose.follow_all_instructions,\n \"inst_level_loose_acc\": out_loose.follow_instruction_list,\n }\n",
27
+ "description": "",
28
+ "target_delimiter": " ",
29
+ "fewshot_delimiter": "\n\n",
30
+ "num_fewshot": 0,
31
+ "metric_list": [
32
+ {
33
+ "metric": "prompt_level_strict_acc",
34
+ "aggregation": "mean",
35
+ "higher_is_better": true
36
+ },
37
+ {
38
+ "metric": "inst_level_strict_acc",
39
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
40
+ "higher_is_better": true
41
+ },
42
+ {
43
+ "metric": "prompt_level_loose_acc",
44
+ "aggregation": "mean",
45
+ "higher_is_better": true
46
+ },
47
+ {
48
+ "metric": "inst_level_loose_acc",
49
+ "aggregation": "def agg_inst_level_acc(items):\n flat_items = [item for sublist in items for item in sublist]\n inst_level_acc = sum(flat_items) / len(flat_items)\n return inst_level_acc\n",
50
+ "higher_is_better": true
51
+ }
52
+ ],
53
+ "output_type": "generate_until",
54
+ "generation_kwargs": {
55
+ "until": [],
56
+ "do_sample": false,
57
+ "temperature": 0.0,
58
+ "max_gen_toks": 1280
59
+ },
60
+ "repeats": 1,
61
+ "should_decontaminate": false,
62
+ "metadata": {
63
+ "version": 4.0,
64
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
65
+ "tensor_parallel_size": 8,
66
+ "dtype": "float32",
67
+ "gpu_memory_utilization": 0.8
68
+ }
69
+ }
70
+ },
71
+ "versions": {
72
+ "ifeval": 4.0
73
+ },
74
+ "n-shot": {
75
+ "ifeval": 0
76
+ },
77
+ "higher_is_better": {
78
+ "ifeval": {
79
+ "prompt_level_strict_acc": true,
80
+ "inst_level_strict_acc": true,
81
+ "prompt_level_loose_acc": true,
82
+ "inst_level_loose_acc": true
83
+ }
84
+ },
85
+ "n-samples": {
86
+ "ifeval": {
87
+ "original": 541,
88
+ "effective": 541
89
+ }
90
+ },
91
+ "config": {
92
+ "model": "vllm",
93
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
94
+ "batch_size": "auto",
95
+ "batch_sizes": [],
96
+ "device": null,
97
+ "use_cache": null,
98
+ "limit": null,
99
+ "bootstrap_iters": 100000,
100
+ "gen_kwargs": null,
101
+ "random_seed": 0,
102
+ "numpy_seed": 1234,
103
+ "torch_seed": 1234,
104
+ "fewshot_seed": 1234
105
+ },
106
+ "git_hash": "18965e2",
107
+ "date": 1754381546.4265583,
108
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 3999.99\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
109
+ "transformers_version": "4.53.0.dev0",
110
+ "lm_eval_version": "0.4.8",
111
+ "upper_git_hash": null,
112
+ "tokenizer_pad_token": [
113
+ "<|end_of_text|>",
114
+ "1"
115
+ ],
116
+ "tokenizer_eos_token": [
117
+ "<|end_of_text|>",
118
+ "1"
119
+ ],
120
+ "tokenizer_bos_token": [
121
+ "<|begin_of_text|>",
122
+ "0"
123
+ ],
124
+ "eot_token_id": 1,
125
+ "max_length": 32768,
126
+ "task_hashes": {
127
+ "ifeval": "a9cc24d7d92904c9f59225bb28b88b892d9ab82be222808ea7fa345ffd4500ae"
128
+ },
129
+ "model_source": "vllm",
130
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
131
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
132
+ "system_instruction": null,
133
+ "system_instruction_sha": null,
134
+ "fewshot_as_multiturn": false,
135
+ "chat_template": null,
136
+ "chat_template_sha": null,
137
+ "start_time": 1214981.212187812,
138
+ "end_time": 1215528.931034912,
139
+ "total_evaluation_time_seconds": "547.7188470999245"
140
+ }
eval_results/ifeval_0shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_ifeval_2025-08-05T08-21-28.685979.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/leaderboard_gpqa_diamond/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T08-31-47.504073.json ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "leaderboard_gpqa_diamond": {
4
+ "alias": "leaderboard_gpqa_diamond",
5
+ "acc_norm,none": 0.3333333333333333,
6
+ "acc_norm_stderr,none": 0.033586181457325226
7
+ }
8
+ },
9
+ "group_subtasks": {
10
+ "leaderboard_gpqa_diamond": []
11
+ },
12
+ "configs": {
13
+ "leaderboard_gpqa_diamond": {
14
+ "task": "leaderboard_gpqa_diamond",
15
+ "dataset_path": "Idavidrein/gpqa",
16
+ "dataset_name": "gpqa_diamond",
17
+ "training_split": "train",
18
+ "validation_split": "train",
19
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n choices = [\n preprocess(doc[\"Incorrect Answer 1\"]),\n preprocess(doc[\"Incorrect Answer 2\"]),\n preprocess(doc[\"Incorrect Answer 3\"]),\n preprocess(doc[\"Correct Answer\"]),\n ]\n\n random.shuffle(choices)\n correct_answer_index = choices.index(preprocess(doc[\"Correct Answer\"]))\n\n out_doc = {\n \"choice1\": choices[0],\n \"choice2\": choices[1],\n \"choice3\": choices[2],\n \"choice4\": choices[3],\n \"answer\": f\"({chr(65 + correct_answer_index)})\",\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
20
+ "doc_to_text": "What is the correct answer to this question:{{Question}}\nChoices:\n(A) {{choice1}}\n(B) {{choice2}}\n(C) {{choice3}}\n(D) {{choice4}}\nAnswer: ",
21
+ "doc_to_target": "answer",
22
+ "unsafe_code": false,
23
+ "doc_to_choice": [
24
+ "(A)",
25
+ "(B)",
26
+ "(C)",
27
+ "(D)"
28
+ ],
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "fewshot_config": {
33
+ "sampler": "first_n"
34
+ },
35
+ "num_fewshot": 0,
36
+ "metric_list": [
37
+ {
38
+ "metric": "acc_norm",
39
+ "aggregation": "mean",
40
+ "higher_is_better": true
41
+ }
42
+ ],
43
+ "output_type": "multiple_choice",
44
+ "repeats": 1,
45
+ "should_decontaminate": false,
46
+ "metadata": {
47
+ "version": 1.0,
48
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
49
+ "tensor_parallel_size": 8,
50
+ "dtype": "float32",
51
+ "gpu_memory_utilization": 0.7
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "leaderboard_gpqa_diamond": 1.0
57
+ },
58
+ "n-shot": {
59
+ "leaderboard_gpqa_diamond": 0
60
+ },
61
+ "higher_is_better": {
62
+ "leaderboard_gpqa_diamond": {
63
+ "acc_norm": true
64
+ }
65
+ },
66
+ "n-samples": {
67
+ "leaderboard_gpqa_diamond": {
68
+ "original": 198,
69
+ "effective": 198
70
+ }
71
+ },
72
+ "config": {
73
+ "model": "vllm",
74
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.7",
75
+ "batch_size": "1",
76
+ "batch_sizes": [],
77
+ "device": null,
78
+ "use_cache": null,
79
+ "limit": null,
80
+ "bootstrap_iters": 100000,
81
+ "gen_kwargs": null,
82
+ "random_seed": 0,
83
+ "numpy_seed": 1234,
84
+ "torch_seed": 1234,
85
+ "fewshot_seed": 1234
86
+ },
87
+ "git_hash": "18965e2",
88
+ "date": 1754382403.8955624,
89
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 3999.99\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
90
+ "transformers_version": "4.53.0.dev0",
91
+ "lm_eval_version": "0.4.8",
92
+ "upper_git_hash": null,
93
+ "tokenizer_pad_token": [
94
+ "<|end_of_text|>",
95
+ "1"
96
+ ],
97
+ "tokenizer_eos_token": [
98
+ "<|end_of_text|>",
99
+ "1"
100
+ ],
101
+ "tokenizer_bos_token": [
102
+ "<|begin_of_text|>",
103
+ "0"
104
+ ],
105
+ "eot_token_id": 1,
106
+ "max_length": 32768,
107
+ "task_hashes": {
108
+ "leaderboard_gpqa_diamond": "45f449f3b3dfc0be532cf1913f1559cf9d7645e56ec3c3fe01317fc575a54e3d"
109
+ },
110
+ "model_source": "vllm",
111
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
112
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
113
+ "system_instruction": null,
114
+ "system_instruction_sha": null,
115
+ "fewshot_as_multiturn": false,
116
+ "chat_template": null,
117
+ "chat_template_sha": null,
118
+ "start_time": 859493.879168647,
119
+ "end_time": 859802.798484168,
120
+ "total_evaluation_time_seconds": "308.91931552102324"
121
+ }
eval_results/leaderboard_gpqa_diamond/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_leaderboard_gpqa_diamond_2025-08-05T08-31-47.504073.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/mbpp_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-05T07-47-32.772547.json ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "mbpp": {
4
+ "alias": "mbpp",
5
+ "pass_at_1,none": 0.594,
6
+ "pass_at_1_stderr,none": 0.021983962090086333
7
+ }
8
+ },
9
+ "group_subtasks": {
10
+ "mbpp": []
11
+ },
12
+ "configs": {
13
+ "mbpp": {
14
+ "task": "mbpp",
15
+ "dataset_path": "google-research-datasets/mbpp",
16
+ "dataset_name": "full",
17
+ "test_split": "test",
18
+ "doc_to_text": "You are an expert Python programmer, and here is your task: {{text}} Your code should pass these tests:\n\n{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}\n[BEGIN]\n",
19
+ "doc_to_target": "{% if is_fewshot is defined %}{{code}}\n[DONE]{% else %}{{test_list[0]}}\n{{test_list[1]}}\n{{test_list[2]}}{% endif %}",
20
+ "unsafe_code": true,
21
+ "description": "",
22
+ "target_delimiter": "",
23
+ "fewshot_delimiter": "\n\n",
24
+ "fewshot_config": {
25
+ "sampler": "first_n",
26
+ "samples": "<function list_fewshot_samples at 0x147eda910720>"
27
+ },
28
+ "num_fewshot": 3,
29
+ "metric_list": [
30
+ {
31
+ "metric": "def pass_at_1(references, predictions):\n return pass_at_k.compute(\n references=references,\n predictions=[predictions],\n k=[1],\n )[0][\"pass@1\"]\n",
32
+ "aggregation": "mean",
33
+ "higher_is_better": true
34
+ }
35
+ ],
36
+ "output_type": "generate_until",
37
+ "generation_kwargs": {
38
+ "until": [
39
+ "[DONE]"
40
+ ],
41
+ "do_sample": false
42
+ },
43
+ "repeats": 1,
44
+ "should_decontaminate": false,
45
+ "metadata": {
46
+ "version": 1.0,
47
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
48
+ "tensor_parallel_size": 8,
49
+ "dtype": "float32",
50
+ "gpu_memory_utilization": 0.8
51
+ }
52
+ }
53
+ },
54
+ "versions": {
55
+ "mbpp": 1.0
56
+ },
57
+ "n-shot": {
58
+ "mbpp": 3
59
+ },
60
+ "higher_is_better": {
61
+ "mbpp": {
62
+ "pass_at_1": true
63
+ }
64
+ },
65
+ "n-samples": {
66
+ "mbpp": {
67
+ "original": 500,
68
+ "effective": 500
69
+ }
70
+ },
71
+ "config": {
72
+ "model": "vllm",
73
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
74
+ "batch_size": "1",
75
+ "batch_sizes": [],
76
+ "device": null,
77
+ "use_cache": null,
78
+ "limit": null,
79
+ "bootstrap_iters": 100000,
80
+ "gen_kwargs": null,
81
+ "random_seed": 0,
82
+ "numpy_seed": 1234,
83
+ "torch_seed": 1234,
84
+ "fewshot_seed": 1234
85
+ },
86
+ "git_hash": "18965e2",
87
+ "date": 1754379279.2352705,
88
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.4 | packaged by Anaconda, Inc. | (main, Jun 18 2024, 15:12:24) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 4000.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
89
+ "transformers_version": "4.53.0.dev0",
90
+ "lm_eval_version": "0.4.8",
91
+ "upper_git_hash": null,
92
+ "tokenizer_pad_token": [
93
+ "<|end_of_text|>",
94
+ "1"
95
+ ],
96
+ "tokenizer_eos_token": [
97
+ "<|end_of_text|>",
98
+ "1"
99
+ ],
100
+ "tokenizer_bos_token": [
101
+ "<|begin_of_text|>",
102
+ "0"
103
+ ],
104
+ "eot_token_id": 1,
105
+ "max_length": 32768,
106
+ "task_hashes": {
107
+ "mbpp": "62356f643ed511c895814c093ddb55451941156ba9febb31dc87c6e00d9723e3"
108
+ },
109
+ "model_source": "vllm",
110
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
111
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
112
+ "system_instruction": null,
113
+ "system_instruction_sha": null,
114
+ "fewshot_as_multiturn": false,
115
+ "chat_template": null,
116
+ "chat_template_sha": null,
117
+ "start_time": 3070521.115272583,
118
+ "end_time": 3071300.016741352,
119
+ "total_evaluation_time_seconds": "778.9014687691815"
120
+ }
eval_results/mbpp_3shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_mbpp_2025-08-05T07-47-32.772547.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/minerva_math_4shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/results_2025-08-14T03-07-36.370756.json ADDED
@@ -0,0 +1,584 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "minerva_math": {
4
+ "math_verify,none": 0.343,
5
+ "math_verify_stderr,none": 0.006249062182626097,
6
+ "exact_match,none": 0.377,
7
+ "exact_match_stderr,none": 0.006358261083312397,
8
+ "alias": "minerva_math"
9
+ },
10
+ "minerva_math_algebra": {
11
+ "alias": " - minerva_math_algebra",
12
+ "exact_match,none": 0.5450716090985678,
13
+ "exact_match_stderr,none": 0.014459589267980832,
14
+ "math_verify,none": 0.5046335299073293,
15
+ "math_verify_stderr,none": 0.01451807416884807
16
+ },
17
+ "minerva_math_counting_and_prob": {
18
+ "alias": " - minerva_math_counting_and_prob",
19
+ "exact_match,none": 0.3080168776371308,
20
+ "exact_match_stderr,none": 0.021227773140159355,
21
+ "math_verify,none": 0.2974683544303797,
22
+ "math_verify_stderr,none": 0.021019518390476322
23
+ },
24
+ "minerva_math_geometry": {
25
+ "alias": " - minerva_math_geometry",
26
+ "exact_match,none": 0.31941544885177453,
27
+ "exact_match_stderr,none": 0.021325786338202517,
28
+ "math_verify,none": 0.24634655532359082,
29
+ "math_verify_stderr,none": 0.0197081175002947
30
+ },
31
+ "minerva_math_intermediate_algebra": {
32
+ "alias": " - minerva_math_intermediate_algebra",
33
+ "exact_match,none": 0.1461794019933555,
34
+ "exact_match_stderr,none": 0.011763136470769866,
35
+ "math_verify,none": 0.12624584717607973,
36
+ "math_verify_stderr,none": 0.011058593855296475
37
+ },
38
+ "minerva_math_num_theory": {
39
+ "alias": " - minerva_math_num_theory",
40
+ "exact_match,none": 0.3351851851851852,
41
+ "exact_match_stderr,none": 0.020332855268557177,
42
+ "math_verify,none": 0.34074074074074073,
43
+ "math_verify_stderr,none": 0.02041483001374619
44
+ },
45
+ "minerva_math_prealgebra": {
46
+ "alias": " - minerva_math_prealgebra",
47
+ "exact_match,none": 0.6211251435132032,
48
+ "exact_match_stderr,none": 0.016446664043846586,
49
+ "math_verify,none": 0.5694603903559128,
50
+ "math_verify_stderr,none": 0.016787216475010247
51
+ },
52
+ "minerva_math_precalc": {
53
+ "alias": " - minerva_math_precalc",
54
+ "exact_match,none": 0.15567765567765568,
55
+ "exact_match_stderr,none": 0.015529913319368257,
56
+ "math_verify,none": 0.11538461538461539,
57
+ "math_verify_stderr,none": 0.013685256643164714
58
+ }
59
+ },
60
+ "groups": {
61
+ "minerva_math": {
62
+ "math_verify,none": 0.343,
63
+ "math_verify_stderr,none": 0.006249062182626097,
64
+ "exact_match,none": 0.377,
65
+ "exact_match_stderr,none": 0.006358261083312397,
66
+ "alias": "minerva_math"
67
+ }
68
+ },
69
+ "group_subtasks": {
70
+ "minerva_math": [
71
+ "minerva_math_algebra",
72
+ "minerva_math_counting_and_prob",
73
+ "minerva_math_geometry",
74
+ "minerva_math_intermediate_algebra",
75
+ "minerva_math_num_theory",
76
+ "minerva_math_prealgebra",
77
+ "minerva_math_precalc"
78
+ ]
79
+ },
80
+ "configs": {
81
+ "minerva_math_algebra": {
82
+ "task": "minerva_math_algebra",
83
+ "tag": [
84
+ "math_word_problems"
85
+ ],
86
+ "dataset_path": "EleutherAI/hendrycks_math",
87
+ "dataset_name": "algebra",
88
+ "training_split": "train",
89
+ "test_split": "test",
90
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
91
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
92
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
93
+ "unsafe_code": false,
94
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
95
+ "description": "",
96
+ "target_delimiter": " ",
97
+ "fewshot_delimiter": "\n\n",
98
+ "fewshot_config": {
99
+ "sampler": "first_n",
100
+ "samples": "<function list_fewshot_samples at 0x14ee488911c0>"
101
+ },
102
+ "num_fewshot": 4,
103
+ "metric_list": [
104
+ {
105
+ "metric": "exact_match",
106
+ "aggregation": "mean",
107
+ "higher_is_better": true
108
+ },
109
+ {
110
+ "metric": "math_verify",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ }
114
+ ],
115
+ "output_type": "generate_until",
116
+ "generation_kwargs": {
117
+ "until": [
118
+ "Problem:"
119
+ ],
120
+ "do_sample": false,
121
+ "temperature": 0.0
122
+ },
123
+ "repeats": 1,
124
+ "should_decontaminate": false,
125
+ "metadata": {
126
+ "version": 2.0,
127
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
128
+ "tensor_parallel_size": 8,
129
+ "dtype": "float32",
130
+ "gpu_memory_utilization": 0.8
131
+ }
132
+ },
133
+ "minerva_math_counting_and_prob": {
134
+ "task": "minerva_math_counting_and_prob",
135
+ "tag": [
136
+ "math_word_problems"
137
+ ],
138
+ "dataset_path": "EleutherAI/hendrycks_math",
139
+ "dataset_name": "counting_and_probability",
140
+ "training_split": "train",
141
+ "test_split": "test",
142
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
143
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
144
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
145
+ "unsafe_code": false,
146
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
147
+ "description": "",
148
+ "target_delimiter": " ",
149
+ "fewshot_delimiter": "\n\n",
150
+ "fewshot_config": {
151
+ "sampler": "first_n",
152
+ "samples": "<function list_fewshot_samples at 0x14ee4887ede0>"
153
+ },
154
+ "num_fewshot": 4,
155
+ "metric_list": [
156
+ {
157
+ "metric": "exact_match",
158
+ "aggregation": "mean",
159
+ "higher_is_better": true
160
+ },
161
+ {
162
+ "metric": "math_verify",
163
+ "aggregation": "mean",
164
+ "higher_is_better": true
165
+ }
166
+ ],
167
+ "output_type": "generate_until",
168
+ "generation_kwargs": {
169
+ "until": [
170
+ "Problem:"
171
+ ],
172
+ "do_sample": false,
173
+ "temperature": 0.0
174
+ },
175
+ "repeats": 1,
176
+ "should_decontaminate": false,
177
+ "metadata": {
178
+ "version": 2.0,
179
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
180
+ "tensor_parallel_size": 8,
181
+ "dtype": "float32",
182
+ "gpu_memory_utilization": 0.8
183
+ }
184
+ },
185
+ "minerva_math_geometry": {
186
+ "task": "minerva_math_geometry",
187
+ "tag": [
188
+ "math_word_problems"
189
+ ],
190
+ "dataset_path": "EleutherAI/hendrycks_math",
191
+ "dataset_name": "geometry",
192
+ "training_split": "train",
193
+ "test_split": "test",
194
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
195
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
196
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
197
+ "unsafe_code": false,
198
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
199
+ "description": "",
200
+ "target_delimiter": " ",
201
+ "fewshot_delimiter": "\n\n",
202
+ "fewshot_config": {
203
+ "sampler": "first_n",
204
+ "samples": "<function list_fewshot_samples at 0x14ee48ccfce0>"
205
+ },
206
+ "num_fewshot": 4,
207
+ "metric_list": [
208
+ {
209
+ "metric": "exact_match",
210
+ "aggregation": "mean",
211
+ "higher_is_better": true
212
+ },
213
+ {
214
+ "metric": "math_verify",
215
+ "aggregation": "mean",
216
+ "higher_is_better": true
217
+ }
218
+ ],
219
+ "output_type": "generate_until",
220
+ "generation_kwargs": {
221
+ "until": [
222
+ "Problem:"
223
+ ],
224
+ "do_sample": false,
225
+ "temperature": 0.0
226
+ },
227
+ "repeats": 1,
228
+ "should_decontaminate": false,
229
+ "metadata": {
230
+ "version": 2.0,
231
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
232
+ "tensor_parallel_size": 8,
233
+ "dtype": "float32",
234
+ "gpu_memory_utilization": 0.8
235
+ }
236
+ },
237
+ "minerva_math_intermediate_algebra": {
238
+ "task": "minerva_math_intermediate_algebra",
239
+ "tag": [
240
+ "math_word_problems"
241
+ ],
242
+ "dataset_path": "EleutherAI/hendrycks_math",
243
+ "dataset_name": "intermediate_algebra",
244
+ "training_split": "train",
245
+ "test_split": "test",
246
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
247
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
248
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
249
+ "unsafe_code": false,
250
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
251
+ "description": "",
252
+ "target_delimiter": " ",
253
+ "fewshot_delimiter": "\n\n",
254
+ "fewshot_config": {
255
+ "sampler": "first_n",
256
+ "samples": "<function list_fewshot_samples at 0x14ee48ccc2c0>"
257
+ },
258
+ "num_fewshot": 4,
259
+ "metric_list": [
260
+ {
261
+ "metric": "exact_match",
262
+ "aggregation": "mean",
263
+ "higher_is_better": true
264
+ },
265
+ {
266
+ "metric": "math_verify",
267
+ "aggregation": "mean",
268
+ "higher_is_better": true
269
+ }
270
+ ],
271
+ "output_type": "generate_until",
272
+ "generation_kwargs": {
273
+ "until": [
274
+ "Problem:"
275
+ ],
276
+ "do_sample": false,
277
+ "temperature": 0.0
278
+ },
279
+ "repeats": 1,
280
+ "should_decontaminate": false,
281
+ "metadata": {
282
+ "version": 2.0,
283
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
284
+ "tensor_parallel_size": 8,
285
+ "dtype": "float32",
286
+ "gpu_memory_utilization": 0.8
287
+ }
288
+ },
289
+ "minerva_math_num_theory": {
290
+ "task": "minerva_math_num_theory",
291
+ "tag": [
292
+ "math_word_problems"
293
+ ],
294
+ "dataset_path": "EleutherAI/hendrycks_math",
295
+ "dataset_name": "number_theory",
296
+ "training_split": "train",
297
+ "test_split": "test",
298
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
299
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
300
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
301
+ "unsafe_code": false,
302
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
303
+ "description": "",
304
+ "target_delimiter": " ",
305
+ "fewshot_delimiter": "\n\n",
306
+ "fewshot_config": {
307
+ "sampler": "first_n",
308
+ "samples": "<function list_fewshot_samples at 0x14ee48c9e700>"
309
+ },
310
+ "num_fewshot": 4,
311
+ "metric_list": [
312
+ {
313
+ "metric": "exact_match",
314
+ "aggregation": "mean",
315
+ "higher_is_better": true
316
+ },
317
+ {
318
+ "metric": "math_verify",
319
+ "aggregation": "mean",
320
+ "higher_is_better": true
321
+ }
322
+ ],
323
+ "output_type": "generate_until",
324
+ "generation_kwargs": {
325
+ "until": [
326
+ "Problem:"
327
+ ],
328
+ "do_sample": false,
329
+ "temperature": 0.0
330
+ },
331
+ "repeats": 1,
332
+ "should_decontaminate": false,
333
+ "metadata": {
334
+ "version": 2.0,
335
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
336
+ "tensor_parallel_size": 8,
337
+ "dtype": "float32",
338
+ "gpu_memory_utilization": 0.8
339
+ }
340
+ },
341
+ "minerva_math_prealgebra": {
342
+ "task": "minerva_math_prealgebra",
343
+ "tag": [
344
+ "math_word_problems"
345
+ ],
346
+ "dataset_path": "EleutherAI/hendrycks_math",
347
+ "dataset_name": "prealgebra",
348
+ "training_split": "train",
349
+ "test_split": "test",
350
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
351
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
352
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
353
+ "unsafe_code": false,
354
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
355
+ "description": "",
356
+ "target_delimiter": " ",
357
+ "fewshot_delimiter": "\n\n",
358
+ "fewshot_config": {
359
+ "sampler": "first_n",
360
+ "samples": "<function list_fewshot_samples at 0x14ee48fd6020>"
361
+ },
362
+ "num_fewshot": 4,
363
+ "metric_list": [
364
+ {
365
+ "metric": "exact_match",
366
+ "aggregation": "mean",
367
+ "higher_is_better": true
368
+ },
369
+ {
370
+ "metric": "math_verify",
371
+ "aggregation": "mean",
372
+ "higher_is_better": true
373
+ }
374
+ ],
375
+ "output_type": "generate_until",
376
+ "generation_kwargs": {
377
+ "until": [
378
+ "Problem:"
379
+ ],
380
+ "do_sample": false,
381
+ "temperature": 0.0
382
+ },
383
+ "repeats": 1,
384
+ "should_decontaminate": false,
385
+ "metadata": {
386
+ "version": 2.0,
387
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
388
+ "tensor_parallel_size": 8,
389
+ "dtype": "float32",
390
+ "gpu_memory_utilization": 0.8
391
+ }
392
+ },
393
+ "minerva_math_precalc": {
394
+ "task": "minerva_math_precalc",
395
+ "tag": [
396
+ "math_word_problems"
397
+ ],
398
+ "dataset_path": "EleutherAI/hendrycks_math",
399
+ "dataset_name": "precalculus",
400
+ "training_split": "train",
401
+ "test_split": "test",
402
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc: dict) -> dict:\n out_doc = {\n \"problem\": doc[\"problem\"],\n \"solution\": doc[\"solution\"],\n \"answer\": normalize_final_answer(\n remove_boxed(last_boxed_only_string(doc[\"solution\"]))\n ),\n }\n if getattr(doc, \"few_shot\", None) is not None:\n out_doc[\"few_shot\"] = True\n return out_doc\n\n return dataset.map(_process_doc)\n",
403
+ "doc_to_text": "def doc_to_text(doc: dict) -> str:\n return \"Problem:\" + \"\\n\" + doc[\"problem\"] + \"\\n\\n\" + \"Solution:\"\n",
404
+ "doc_to_target": "{{answer if few_shot is undefined else solution}}",
405
+ "unsafe_code": false,
406
+ "process_results": "def process_results(doc: dict, results: List[str]) -> Dict[str, int]:\n candidates = results[0]\n\n unnormalized_answer = get_unnormalized_answer(candidates)\n answer = normalize_final_answer(unnormalized_answer)\n\n if is_equiv(answer, doc[\"answer\"]):\n retval = 1\n else:\n retval = 0\n\n # math_verify\n res = verify(parse(doc[\"answer\"]), parse(candidates))\n mathval = 1 if res else 0\n\n results = {\n \"exact_match\": retval,\n \"math_verify\": mathval,\n }\n return results\n",
407
+ "description": "",
408
+ "target_delimiter": " ",
409
+ "fewshot_delimiter": "\n\n",
410
+ "fewshot_config": {
411
+ "sampler": "first_n",
412
+ "samples": "<function list_fewshot_samples at 0x14ee48fd68e0>"
413
+ },
414
+ "num_fewshot": 4,
415
+ "metric_list": [
416
+ {
417
+ "metric": "exact_match",
418
+ "aggregation": "mean",
419
+ "higher_is_better": true
420
+ },
421
+ {
422
+ "metric": "math_verify",
423
+ "aggregation": "mean",
424
+ "higher_is_better": true
425
+ }
426
+ ],
427
+ "output_type": "generate_until",
428
+ "generation_kwargs": {
429
+ "until": [
430
+ "Problem:"
431
+ ],
432
+ "do_sample": false,
433
+ "temperature": 0.0
434
+ },
435
+ "repeats": 1,
436
+ "should_decontaminate": false,
437
+ "metadata": {
438
+ "version": 2.0,
439
+ "pretrained": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
440
+ "tensor_parallel_size": 8,
441
+ "dtype": "float32",
442
+ "gpu_memory_utilization": 0.8
443
+ }
444
+ }
445
+ },
446
+ "versions": {
447
+ "minerva_math": 1.0,
448
+ "minerva_math_algebra": 2.0,
449
+ "minerva_math_counting_and_prob": 2.0,
450
+ "minerva_math_geometry": 2.0,
451
+ "minerva_math_intermediate_algebra": 2.0,
452
+ "minerva_math_num_theory": 2.0,
453
+ "minerva_math_prealgebra": 2.0,
454
+ "minerva_math_precalc": 2.0
455
+ },
456
+ "n-shot": {
457
+ "minerva_math_algebra": 4,
458
+ "minerva_math_counting_and_prob": 4,
459
+ "minerva_math_geometry": 4,
460
+ "minerva_math_intermediate_algebra": 4,
461
+ "minerva_math_num_theory": 4,
462
+ "minerva_math_prealgebra": 4,
463
+ "minerva_math_precalc": 4
464
+ },
465
+ "higher_is_better": {
466
+ "minerva_math": {
467
+ "exact_match": true,
468
+ "math_verify": true
469
+ },
470
+ "minerva_math_algebra": {
471
+ "exact_match": true,
472
+ "math_verify": true
473
+ },
474
+ "minerva_math_counting_and_prob": {
475
+ "exact_match": true,
476
+ "math_verify": true
477
+ },
478
+ "minerva_math_geometry": {
479
+ "exact_match": true,
480
+ "math_verify": true
481
+ },
482
+ "minerva_math_intermediate_algebra": {
483
+ "exact_match": true,
484
+ "math_verify": true
485
+ },
486
+ "minerva_math_num_theory": {
487
+ "exact_match": true,
488
+ "math_verify": true
489
+ },
490
+ "minerva_math_prealgebra": {
491
+ "exact_match": true,
492
+ "math_verify": true
493
+ },
494
+ "minerva_math_precalc": {
495
+ "exact_match": true,
496
+ "math_verify": true
497
+ }
498
+ },
499
+ "n-samples": {
500
+ "minerva_math_algebra": {
501
+ "original": 1187,
502
+ "effective": 1187
503
+ },
504
+ "minerva_math_counting_and_prob": {
505
+ "original": 474,
506
+ "effective": 474
507
+ },
508
+ "minerva_math_geometry": {
509
+ "original": 479,
510
+ "effective": 479
511
+ },
512
+ "minerva_math_intermediate_algebra": {
513
+ "original": 903,
514
+ "effective": 903
515
+ },
516
+ "minerva_math_num_theory": {
517
+ "original": 540,
518
+ "effective": 540
519
+ },
520
+ "minerva_math_prealgebra": {
521
+ "original": 871,
522
+ "effective": 871
523
+ },
524
+ "minerva_math_precalc": {
525
+ "original": 546,
526
+ "effective": 546
527
+ }
528
+ },
529
+ "config": {
530
+ "model": "vllm",
531
+ "model_args": "pretrained=/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000,tensor_parallel_size=8,dtype=float32,gpu_memory_utilization=0.8",
532
+ "batch_size": "auto",
533
+ "batch_sizes": [],
534
+ "device": null,
535
+ "use_cache": null,
536
+ "limit": null,
537
+ "bootstrap_iters": 100000,
538
+ "gen_kwargs": null,
539
+ "random_seed": 0,
540
+ "numpy_seed": 1234,
541
+ "torch_seed": 1234,
542
+ "fewshot_seed": 1234
543
+ },
544
+ "git_hash": "18965e2",
545
+ "date": 1755138553.1148791,
546
+ "pretty_env_info": "PyTorch version: 2.7.0+cu126\nIs debug build: False\nCUDA used to build PyTorch: 12.6\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.7\nLibc version: glibc-2.35\n\nPython version: 3.12.2 | packaged by conda-forge | (main, Feb 16 2024, 20:50:58) [GCC 12.3.0] (64-bit runtime)\nPython platform: Linux-5.15.0-1088-azure-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA H200\nGPU 1: NVIDIA H200\nGPU 2: NVIDIA H200\nGPU 3: NVIDIA H200\nGPU 4: NVIDIA H200\nGPU 5: NVIDIA H200\nGPU 6: NVIDIA H200\nGPU 7: NVIDIA H200\n\nNvidia driver version: 570.133.20\ncuDNN version: Could not collect\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 96\nOn-line CPU(s) list: 0-95\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8480C\nCPU family: 6\nModel: 143\nThread(s) per core: 1\nCore(s) per socket: 48\nSocket(s): 2\nStepping: 8\nBogoMIPS: 3999.99\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology tsc_reliable nonstop_tsc cpuid aperfmperf tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch invpcid_single fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves avx_vnni avx512_bf16 avx512vbmi umip waitpkg avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg avx512_vpopcntdq la57 rdpid cldemote movdiri movdir64b fsrm serialize amx_bf16 avx512_fp16 amx_tile amx_int8 arch_capabilities\nHypervisor vendor: Microsoft\nVirtualization type: full\nL1d cache: 4.5 MiB (96 instances)\nL1i cache: 3 MiB (96 instances)\nL2 cache: 192 MiB (96 instances)\nL3 cache: 210 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-47\nNUMA node1 CPU(s): 48-95\nVulnerability Gather data sampling: Not affected\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Mmio stale data: Unknown: No mitigations\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Retpolines; STIBP disabled; RSB filling; PBRSB-eIBRS Not affected; BHI Retpoline\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] flake8==7.0.0\n[pip3] flashinfer-python==0.2.5+cu126torch2.6\n[pip3] mypy==1.10.0\n[pip3] mypy-extensions==1.0.0\n[pip3] numpy==1.26.4\n[pip3] numpydoc==1.7.0\n[pip3] nvidia-cublas-cu12==12.6.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.6.80\n[pip3] nvidia-cuda-nvrtc-cu12==12.6.77\n[pip3] nvidia-cuda-runtime-cu12==12.6.77\n[pip3] nvidia-cudnn-cu12==9.5.1.17\n[pip3] nvidia-cufft-cu12==11.3.0.4\n[pip3] nvidia-curand-cu12==10.3.7.77\n[pip3] nvidia-cusolver-cu12==11.7.1.2\n[pip3] nvidia-cusparse-cu12==12.5.4.2\n[pip3] nvidia-cusparselt-cu12==0.6.3\n[pip3] nvidia-nccl-cu12==2.26.2\n[pip3] nvidia-nvjitlink-cu12==12.6.85\n[pip3] nvidia-nvtx-cu12==12.6.77\n[pip3] torch==2.7.0\n[pip3] torchaudio==2.7.0\n[pip3] torchvision==0.22.0\n[pip3] triton==3.3.0\n[conda] _anaconda_depends 2024.06 py312_mkl_2 \n[conda] blas 1.0 mkl \n[conda] flashinfer-python 0.2.5+cu126torch2.6 pypi_0 pypi\n[conda] mkl 2023.1.0 h213fc3f_46344 \n[conda] mkl-service 2.4.0 py312h5eee18b_1 \n[conda] mkl_fft 1.3.8 py312h5eee18b_0 \n[conda] mkl_random 1.2.4 py312hdb19cb5_0 \n[conda] numpy 1.26.4 py312hc5e2394_0 \n[conda] numpy-base 1.26.4 py312h0da6c21_0 \n[conda] numpydoc 1.7.0 py312h06a4308_0 \n[conda] nvidia-cublas-cu12 12.6.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.6.80 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.6.77 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.5.1.17 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.0.4 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.7.77 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.1.2 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.4.2 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.6.3 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.26.2 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.6.85 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.6.77 pypi_0 pypi\n[conda] torch 2.7.0 pypi_0 pypi\n[conda] torchaudio 2.7.0 pypi_0 pypi\n[conda] torchvision 0.22.0 pypi_0 pypi\n[conda] triton 3.3.0 pypi_0 pypi",
547
+ "transformers_version": "4.53.0.dev0",
548
+ "lm_eval_version": "0.4.9.1",
549
+ "upper_git_hash": null,
550
+ "tokenizer_pad_token": [
551
+ "<|end_of_text|>",
552
+ "1"
553
+ ],
554
+ "tokenizer_eos_token": [
555
+ "<|end_of_text|>",
556
+ "1"
557
+ ],
558
+ "tokenizer_bos_token": [
559
+ "<|begin_of_text|>",
560
+ "0"
561
+ ],
562
+ "eot_token_id": 1,
563
+ "max_length": 32768,
564
+ "task_hashes": {
565
+ "minerva_math_algebra": "5c955bbc89ad645142d61b1594b7c36b552b722edf416ae40fcc71a4c50bd24b",
566
+ "minerva_math_counting_and_prob": "44b9697d6c9aa5b4c364a427ece31698d9eb853f35b2b059c11a461b8886534e",
567
+ "minerva_math_geometry": "e3bc2da59c734f3345ac1db47104b32ddcaf82e460a2dc3449e2c88249e4e1fb",
568
+ "minerva_math_intermediate_algebra": "fba9ce144ffb78d824e4e4cc707e887c24afd73cc95ae48c38feef96e61fc77c",
569
+ "minerva_math_num_theory": "a54599f16065edfa4a097d2e6d0c7f71d92ece79ff5d4910abcc374456f6b352",
570
+ "minerva_math_prealgebra": "9d0a86e21bfe1ffa07f634fec45d83c27d6190dd7b452230e405b7640a28fd6f",
571
+ "minerva_math_precalc": "77e35064ebbe841cd39c111b65213ee245825d611c4bf7920b08c823d8db65ef"
572
+ },
573
+ "model_source": "vllm",
574
+ "model_name": "/lustrefs/users/runner/workspace/checkpoints/huggingface/k2plus_stage1_attn8k_jais250k_tp8/checkpoints/checkpoint_0065000",
575
+ "model_name_sanitized": "__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000",
576
+ "system_instruction": null,
577
+ "system_instruction_sha": null,
578
+ "fewshot_as_multiturn": false,
579
+ "chat_template": null,
580
+ "chat_template_sha": null,
581
+ "start_time": 3829775.640661165,
582
+ "end_time": 3832097.2234475,
583
+ "total_evaluation_time_seconds": "2321.5827863346785"
584
+ }
eval_results/minerva_math_4shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_minerva_math_algebra_2025-08-14T03-07-36.370756.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
eval_results/minerva_math_4shots/__lustrefs__users__runner__workspace__checkpoints__huggingface__k2plus_stage1_attn8k_jais250k_tp8__checkpoints__checkpoint_0065000/samples_minerva_math_counting_and_prob_2025-08-14T03-07-36.370756.jsonl ADDED
The diff for this file is too large to render. See raw diff