Run 4. Outer Step 90. Inner Step 0. Peers 86.
Browse files- config.json +24 -24
- inner_optimizer.pt +1 -1
- model.safetensors +1 -1
- outer_optimizer.pt +1 -1
config.json
CHANGED
@@ -12,14 +12,14 @@
|
|
12 |
"104": "NON_PARTICIPATING",
|
13 |
"105": "NON_PARTICIPATING",
|
14 |
"106": "NON_PARTICIPATING",
|
15 |
-
"107": "
|
16 |
"108": "NON_PARTICIPATING",
|
17 |
"109": "NON_PARTICIPATING",
|
18 |
"11": "NON_PARTICIPATING",
|
19 |
"110": "NON_PARTICIPATING",
|
20 |
"111": "NON_PARTICIPATING",
|
21 |
"112": "NON_PARTICIPATING",
|
22 |
-
"113": "
|
23 |
"114": "NON_PARTICIPATING",
|
24 |
"115": "NON_PARTICIPATING",
|
25 |
"116": "NON_PARTICIPATING",
|
@@ -35,15 +35,15 @@
|
|
35 |
"125": "NON_PARTICIPATING",
|
36 |
"126": "NON_PARTICIPATING",
|
37 |
"127": "NON_PARTICIPATING",
|
38 |
-
"128": "
|
39 |
-
"129": "
|
40 |
"13": "SUCCESS",
|
41 |
-
"130": "
|
42 |
"131": "NON_PARTICIPATING",
|
43 |
"132": "NON_PARTICIPATING",
|
44 |
-
"133": "
|
45 |
"134": "NON_PARTICIPATING",
|
46 |
-
"135": "
|
47 |
"136": "NON_PARTICIPATING",
|
48 |
"137": "NON_PARTICIPATING",
|
49 |
"138": "NON_PARTICIPATING",
|
@@ -60,7 +60,7 @@
|
|
60 |
"148": "NON_PARTICIPATING",
|
61 |
"149": "NON_PARTICIPATING",
|
62 |
"15": "SUCCESS",
|
63 |
-
"150": "
|
64 |
"151": "NON_PARTICIPATING",
|
65 |
"152": "NON_PARTICIPATING",
|
66 |
"153": "NON_PARTICIPATING",
|
@@ -72,10 +72,10 @@
|
|
72 |
"159": "NON_PARTICIPATING",
|
73 |
"16": "SUCCESS",
|
74 |
"160": "NON_PARTICIPATING",
|
75 |
-
"161": "
|
76 |
"162": "NON_PARTICIPATING",
|
77 |
"163": "NON_PARTICIPATING",
|
78 |
-
"164": "
|
79 |
"165": "NON_PARTICIPATING",
|
80 |
"166": "NON_PARTICIPATING",
|
81 |
"167": "NON_PARTICIPATING",
|
@@ -94,10 +94,10 @@
|
|
94 |
"179": "NON_PARTICIPATING",
|
95 |
"18": "SUCCESS",
|
96 |
"180": "NON_PARTICIPATING",
|
97 |
-
"181": "
|
98 |
"182": "NON_PARTICIPATING",
|
99 |
"183": "NON_PARTICIPATING",
|
100 |
-
"184": "
|
101 |
"185": "NON_PARTICIPATING",
|
102 |
"186": "NON_PARTICIPATING",
|
103 |
"187": "NON_PARTICIPATING",
|
@@ -110,10 +110,10 @@
|
|
110 |
"193": "NON_PARTICIPATING",
|
111 |
"194": "NON_PARTICIPATING",
|
112 |
"195": "NON_PARTICIPATING",
|
113 |
-
"196": "
|
114 |
"197": "NON_PARTICIPATING",
|
115 |
"198": "NON_PARTICIPATING",
|
116 |
-
"199": "
|
117 |
"2": "SUCCESS",
|
118 |
"20": "SUCCESS",
|
119 |
"200": "NON_PARTICIPATING",
|
@@ -133,8 +133,8 @@
|
|
133 |
"213": "NON_PARTICIPATING",
|
134 |
"214": "NON_PARTICIPATING",
|
135 |
"215": "NON_PARTICIPATING",
|
136 |
-
"216": "
|
137 |
-
"217": "
|
138 |
"218": "NON_PARTICIPATING",
|
139 |
"219": "NON_PARTICIPATING",
|
140 |
"22": "SUCCESS",
|
@@ -145,15 +145,15 @@
|
|
145 |
"224": "NON_PARTICIPATING",
|
146 |
"225": "NON_PARTICIPATING",
|
147 |
"226": "NON_PARTICIPATING",
|
148 |
-
"227": "
|
149 |
"228": "NON_PARTICIPATING",
|
150 |
"229": "NON_PARTICIPATING",
|
151 |
"23": "NON_PARTICIPATING",
|
152 |
"230": "NON_PARTICIPATING",
|
153 |
"231": "NON_PARTICIPATING",
|
154 |
"232": "NON_PARTICIPATING",
|
155 |
-
"233": "
|
156 |
-
"234": "
|
157 |
"235": "NON_PARTICIPATING",
|
158 |
"236": "NON_PARTICIPATING",
|
159 |
"237": "NON_PARTICIPATING",
|
@@ -163,7 +163,7 @@
|
|
163 |
"240": "NON_PARTICIPATING",
|
164 |
"241": "NON_PARTICIPATING",
|
165 |
"242": "NON_PARTICIPATING",
|
166 |
-
"243": "
|
167 |
"244": "NON_PARTICIPATING",
|
168 |
"245": "NON_PARTICIPATING",
|
169 |
"246": "NON_PARTICIPATING",
|
@@ -171,13 +171,13 @@
|
|
171 |
"248": "NON_PARTICIPATING",
|
172 |
"249": "NON_PARTICIPATING",
|
173 |
"25": "SUCCESS",
|
174 |
-
"250": "
|
175 |
"251": "NON_PARTICIPATING",
|
176 |
"252": "NON_PARTICIPATING",
|
177 |
"253": "NON_PARTICIPATING",
|
178 |
"254": "NON_PARTICIPATING",
|
179 |
"255": "NON_PARTICIPATING",
|
180 |
-
"26": "
|
181 |
"27": "SUCCESS",
|
182 |
"28": "SUCCESS",
|
183 |
"29": "SUCCESS",
|
@@ -225,7 +225,7 @@
|
|
225 |
"67": "SUCCESS",
|
226 |
"68": "NON_PARTICIPATING",
|
227 |
"69": "NON_PARTICIPATING",
|
228 |
-
"7": "
|
229 |
"70": "SUCCESS",
|
230 |
"71": "SUCCESS",
|
231 |
"72": "SUCCESS",
|
@@ -275,7 +275,7 @@
|
|
275 |
"initializer_range": 0.02,
|
276 |
"inner_step": 0,
|
277 |
"inner_steps": 0,
|
278 |
-
"last_allreduce_block":
|
279 |
"layer_norm_epsilon": 1e-05,
|
280 |
"model_type": "gpt_optimized",
|
281 |
"n_embd": 1280,
|
|
|
12 |
"104": "NON_PARTICIPATING",
|
13 |
"105": "NON_PARTICIPATING",
|
14 |
"106": "NON_PARTICIPATING",
|
15 |
+
"107": "SUCCESS",
|
16 |
"108": "NON_PARTICIPATING",
|
17 |
"109": "NON_PARTICIPATING",
|
18 |
"11": "NON_PARTICIPATING",
|
19 |
"110": "NON_PARTICIPATING",
|
20 |
"111": "NON_PARTICIPATING",
|
21 |
"112": "NON_PARTICIPATING",
|
22 |
+
"113": "NON_PARTICIPATING",
|
23 |
"114": "NON_PARTICIPATING",
|
24 |
"115": "NON_PARTICIPATING",
|
25 |
"116": "NON_PARTICIPATING",
|
|
|
35 |
"125": "NON_PARTICIPATING",
|
36 |
"126": "NON_PARTICIPATING",
|
37 |
"127": "NON_PARTICIPATING",
|
38 |
+
"128": "SUCCESS",
|
39 |
+
"129": "NON_PARTICIPATING",
|
40 |
"13": "SUCCESS",
|
41 |
+
"130": "NON_PARTICIPATING",
|
42 |
"131": "NON_PARTICIPATING",
|
43 |
"132": "NON_PARTICIPATING",
|
44 |
+
"133": "SUCCESS",
|
45 |
"134": "NON_PARTICIPATING",
|
46 |
+
"135": "NON_PARTICIPATING",
|
47 |
"136": "NON_PARTICIPATING",
|
48 |
"137": "NON_PARTICIPATING",
|
49 |
"138": "NON_PARTICIPATING",
|
|
|
60 |
"148": "NON_PARTICIPATING",
|
61 |
"149": "NON_PARTICIPATING",
|
62 |
"15": "SUCCESS",
|
63 |
+
"150": "SUCCESS",
|
64 |
"151": "NON_PARTICIPATING",
|
65 |
"152": "NON_PARTICIPATING",
|
66 |
"153": "NON_PARTICIPATING",
|
|
|
72 |
"159": "NON_PARTICIPATING",
|
73 |
"16": "SUCCESS",
|
74 |
"160": "NON_PARTICIPATING",
|
75 |
+
"161": "SUCCESS",
|
76 |
"162": "NON_PARTICIPATING",
|
77 |
"163": "NON_PARTICIPATING",
|
78 |
+
"164": "NON_PARTICIPATING",
|
79 |
"165": "NON_PARTICIPATING",
|
80 |
"166": "NON_PARTICIPATING",
|
81 |
"167": "NON_PARTICIPATING",
|
|
|
94 |
"179": "NON_PARTICIPATING",
|
95 |
"18": "SUCCESS",
|
96 |
"180": "NON_PARTICIPATING",
|
97 |
+
"181": "NON_PARTICIPATING",
|
98 |
"182": "NON_PARTICIPATING",
|
99 |
"183": "NON_PARTICIPATING",
|
100 |
+
"184": "SUCCESS",
|
101 |
"185": "NON_PARTICIPATING",
|
102 |
"186": "NON_PARTICIPATING",
|
103 |
"187": "NON_PARTICIPATING",
|
|
|
110 |
"193": "NON_PARTICIPATING",
|
111 |
"194": "NON_PARTICIPATING",
|
112 |
"195": "NON_PARTICIPATING",
|
113 |
+
"196": "SUCCESS",
|
114 |
"197": "NON_PARTICIPATING",
|
115 |
"198": "NON_PARTICIPATING",
|
116 |
+
"199": "NON_PARTICIPATING",
|
117 |
"2": "SUCCESS",
|
118 |
"20": "SUCCESS",
|
119 |
"200": "NON_PARTICIPATING",
|
|
|
133 |
"213": "NON_PARTICIPATING",
|
134 |
"214": "NON_PARTICIPATING",
|
135 |
"215": "NON_PARTICIPATING",
|
136 |
+
"216": "SUCCESS",
|
137 |
+
"217": "SUCCESS",
|
138 |
"218": "NON_PARTICIPATING",
|
139 |
"219": "NON_PARTICIPATING",
|
140 |
"22": "SUCCESS",
|
|
|
145 |
"224": "NON_PARTICIPATING",
|
146 |
"225": "NON_PARTICIPATING",
|
147 |
"226": "NON_PARTICIPATING",
|
148 |
+
"227": "SUCCESS",
|
149 |
"228": "NON_PARTICIPATING",
|
150 |
"229": "NON_PARTICIPATING",
|
151 |
"23": "NON_PARTICIPATING",
|
152 |
"230": "NON_PARTICIPATING",
|
153 |
"231": "NON_PARTICIPATING",
|
154 |
"232": "NON_PARTICIPATING",
|
155 |
+
"233": "SUCCESS",
|
156 |
+
"234": "NON_PARTICIPATING",
|
157 |
"235": "NON_PARTICIPATING",
|
158 |
"236": "NON_PARTICIPATING",
|
159 |
"237": "NON_PARTICIPATING",
|
|
|
163 |
"240": "NON_PARTICIPATING",
|
164 |
"241": "NON_PARTICIPATING",
|
165 |
"242": "NON_PARTICIPATING",
|
166 |
+
"243": "NON_PARTICIPATING",
|
167 |
"244": "NON_PARTICIPATING",
|
168 |
"245": "NON_PARTICIPATING",
|
169 |
"246": "NON_PARTICIPATING",
|
|
|
171 |
"248": "NON_PARTICIPATING",
|
172 |
"249": "NON_PARTICIPATING",
|
173 |
"25": "SUCCESS",
|
174 |
+
"250": "SUCCESS",
|
175 |
"251": "NON_PARTICIPATING",
|
176 |
"252": "NON_PARTICIPATING",
|
177 |
"253": "NON_PARTICIPATING",
|
178 |
"254": "NON_PARTICIPATING",
|
179 |
"255": "NON_PARTICIPATING",
|
180 |
+
"26": "SUCCESS",
|
181 |
"27": "SUCCESS",
|
182 |
"28": "SUCCESS",
|
183 |
"29": "SUCCESS",
|
|
|
225 |
"67": "SUCCESS",
|
226 |
"68": "NON_PARTICIPATING",
|
227 |
"69": "NON_PARTICIPATING",
|
228 |
+
"7": "NON_PARTICIPATING",
|
229 |
"70": "SUCCESS",
|
230 |
"71": "SUCCESS",
|
231 |
"72": "SUCCESS",
|
|
|
275 |
"initializer_range": 0.02,
|
276 |
"inner_step": 0,
|
277 |
"inner_steps": 0,
|
278 |
+
"last_allreduce_block": 5682191,
|
279 |
"layer_norm_epsilon": 1e-05,
|
280 |
"model_type": "gpt_optimized",
|
281 |
"n_embd": 1280,
|
inner_optimizer.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 8081782026
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:6c93d8fae4119f97d5168d07548b9f608c149f2d7dd5a755dff04e1242b6aaa9
|
3 |
size 8081782026
|
model.safetensors
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 4040701744
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:ab90de0036f6726e7aa47e787eb7c1cc0dbf93eac19ce6f29654ab16318cc5f8
|
3 |
size 4040701744
|
outer_optimizer.pt
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 4040805354
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:b8bff2f867ef0390a9b229496960ecfd1ac0cd9a45589fc01dc7c888be857a73
|
3 |
size 4040805354
|