Shasidharyadavr commited on
Commit
54493d9
·
verified ·
1 Parent(s): 2fb3afc

GRPO 3B v8 (temp=0.8, LR=5e-05, r=64, model=Qwen2.5-3B-Instruct): 0.608->0.447 delta=-0.161

Browse files
Files changed (1) hide show
  1. training_results_grpo_3b_v8.json +164 -0
training_results_grpo_3b_v8.json ADDED
@@ -0,0 +1,164 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "env_url": "https://shasidharyadavr-netweaver-sre.hf.space",
3
+ "model_name": "Qwen/Qwen2.5-3B-Instruct",
4
+ "training_method": "GRPO v8: 4-bit + LoRA r=64 (q_proj+k_proj+v_proj+o_proj+gate_proj+up_proj+down_proj), LR=5e-05, temp_train=temp_eval=0.8",
5
+ "max_train_steps": 100,
6
+ "num_generations": 4,
7
+ "learning_rate": 5e-05,
8
+ "lora_rank": 64,
9
+ "lora_target_modules": [
10
+ "q_proj",
11
+ "k_proj",
12
+ "v_proj",
13
+ "o_proj",
14
+ "gate_proj",
15
+ "up_proj",
16
+ "down_proj"
17
+ ],
18
+ "training_temperature": 0.8,
19
+ "eval_temperature": 0.8,
20
+ "trainable_params": 119734272,
21
+ "total_params": 1818406912,
22
+ "training_rewards": [
23
+ 0.9,
24
+ 0.98,
25
+ 0.999,
26
+ 0.999,
27
+ 0.9,
28
+ 0.86,
29
+ 0.3,
30
+ 0.88,
31
+ 0.86,
32
+ 0.9,
33
+ 0.6,
34
+ 0.84,
35
+ 0.9,
36
+ 0.84,
37
+ 0.667,
38
+ 0.84,
39
+ 0.733,
40
+ 0.84,
41
+ 0.82,
42
+ 0.667,
43
+ 0.5,
44
+ 0.7,
45
+ 0.3,
46
+ 0.84,
47
+ 0.3,
48
+ 0.7,
49
+ 0.86,
50
+ 0.8,
51
+ 0.84,
52
+ 0.88,
53
+ 0.86,
54
+ 0.6,
55
+ 0.88,
56
+ 0.667,
57
+ 0.5,
58
+ 0.88,
59
+ 0.5,
60
+ 0.5,
61
+ 0.82,
62
+ 0.9,
63
+ 0.999,
64
+ 0.8,
65
+ 0.999,
66
+ 0.3,
67
+ 0.3,
68
+ 0.3,
69
+ 0.999,
70
+ 0.4,
71
+ 0.86,
72
+ 0.999,
73
+ 0.9,
74
+ 0.86,
75
+ 0.82,
76
+ 0.8,
77
+ 0.88,
78
+ 0.84,
79
+ 0.86,
80
+ 0.88,
81
+ 0.88,
82
+ 0.82,
83
+ 0.84,
84
+ 0.88,
85
+ 0.82,
86
+ 0.84,
87
+ 0.4,
88
+ 0.733,
89
+ 0.86,
90
+ 0.733,
91
+ 0.86,
92
+ 0.98,
93
+ 0.82,
94
+ 0.999,
95
+ 0.999,
96
+ 0.88,
97
+ 0.999,
98
+ 0.999,
99
+ 0.86,
100
+ 0.86,
101
+ 0.6,
102
+ 0.86,
103
+ 0.82,
104
+ 0.84,
105
+ 0.84,
106
+ 0.6,
107
+ 0.86,
108
+ 0.999,
109
+ 0.82,
110
+ 0.5,
111
+ 0.86,
112
+ 0.84,
113
+ 0.86,
114
+ 0.84,
115
+ 0.88,
116
+ 0.98,
117
+ 0.8,
118
+ 0.88,
119
+ 0.88,
120
+ 0.82,
121
+ 0.88,
122
+ 0.94
123
+ ],
124
+ "before_rewards": [
125
+ 0.999,
126
+ 0.3,
127
+ 0.86,
128
+ 0.86,
129
+ 0.3,
130
+ 0.3,
131
+ 0.4,
132
+ 0.88,
133
+ 0.3,
134
+ 0.88
135
+ ],
136
+ "after_rewards": [
137
+ 0.667,
138
+ 0.3,
139
+ 0.5,
140
+ 0.5,
141
+ 0.3,
142
+ 0.3,
143
+ 0.5,
144
+ 0.3,
145
+ 0.6,
146
+ 0.5
147
+ ],
148
+ "difficulty_breakdown": {
149
+ "before": {
150
+ "easy": 0.87,
151
+ "medium": 0.49333333333333335,
152
+ "hard": 0.5718
153
+ },
154
+ "after": {
155
+ "easy": 0.4,
156
+ "medium": 0.3666666666666667,
157
+ "hard": 0.589
158
+ }
159
+ },
160
+ "timestamp": "2026-04-26 07:37:07",
161
+ "source": "grpo_3b_real_run_v8",
162
+ "job_flavor": "a10g-small",
163
+ "notes": "GRPO v8 on Qwen/Qwen2.5-3B-Instruct (4-bit + LoRA r=64, targets=['q_proj', 'k_proj', 'v_proj', 'o_proj', 'gate_proj', 'up_proj', 'down_proj']) vs the live NetWeaver SRE env. Train temp = eval temp = 0.8, LR = 5e-05. Heuristic baseline preserved (separate filenames)."
164
+ }