av-codes commited on
Commit
3d49583
·
verified ·
1 Parent(s): 99486fa

Upload folder using huggingface_hub

Browse files
v3_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "vocab_size": 128,
3
+ "dim": 256,
4
+ "n_layers": 7,
5
+ "n_heads": 4,
6
+ "n_kv_heads": 1,
7
+ "intermediate_size": 704,
8
+ "max_seq_len": 256,
9
+ "rope_theta": 10000.0,
10
+ "rms_norm_eps": 1e-06,
11
+ "dropout": 0.1,
12
+ "tie_weights": true
13
+ }
v3_result.json ADDED
@@ -0,0 +1,296 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "params": 4968192,
3
+ "steps": 20000,
4
+ "elapsed_sec": 1311.850316286087,
5
+ "steps_per_sec": 15.245641786801551,
6
+ "tokens_per_sec": 249784.5950349566,
7
+ "final_train_loss": 0.7182860881090164,
8
+ "best_val_loss": 1.4004173874855042,
9
+ "device": "cuda",
10
+ "data": {
11
+ "cosmopedia": {
12
+ "seen": 318197,
13
+ "kept": 70673,
14
+ "tokens": 14000079,
15
+ "too_short": 698,
16
+ "too_long": 426628,
17
+ "high_unk": 3885,
18
+ "duplicate": 439,
19
+ "off_topic": 157885,
20
+ "stub_dropped": 5
21
+ },
22
+ "science_facts": {
23
+ "seen": 43080,
24
+ "kept": 41649,
25
+ "tokens": 3817837,
26
+ "too_short": 0,
27
+ "too_long": 886,
28
+ "high_unk": 545,
29
+ "duplicate": 0,
30
+ "off_topic": 0,
31
+ "stub_dropped": 0
32
+ },
33
+ "simple_facts": {
34
+ "seen": 4032,
35
+ "kept": 3464,
36
+ "tokens": 182276,
37
+ "too_short": 0,
38
+ "too_long": 52,
39
+ "high_unk": 108,
40
+ "duplicate": 404,
41
+ "off_topic": 0,
42
+ "stub_dropped": 4
43
+ },
44
+ "simplewiki_core": {
45
+ "seen": 106117,
46
+ "kept": 12031,
47
+ "tokens": 2000007,
48
+ "too_short": 23127,
49
+ "too_long": 8107,
50
+ "high_unk": 107,
51
+ "duplicate": 33,
52
+ "off_topic": 61476,
53
+ "stub_dropped": 1236
54
+ },
55
+ "total_tokens": 20000199,
56
+ "total_units_kept": 127817
57
+ },
58
+ "tokenizer": {
59
+ "version": "tiny-qwen-v2-boundary-128",
60
+ "vocab": [
61
+ "<pad>",
62
+ "<unk>",
63
+ "<bos>",
64
+ "<eos>",
65
+ " ",
66
+ "a",
67
+ "b",
68
+ "c",
69
+ "d",
70
+ "e",
71
+ "f",
72
+ "g",
73
+ "h",
74
+ "i",
75
+ "j",
76
+ "k",
77
+ "l",
78
+ "m",
79
+ "n",
80
+ "o",
81
+ "p",
82
+ "q",
83
+ "r",
84
+ "s",
85
+ "t",
86
+ "u",
87
+ "v",
88
+ "w",
89
+ "x",
90
+ "y",
91
+ "z",
92
+ "0",
93
+ "1",
94
+ "2",
95
+ "3",
96
+ "4",
97
+ "5",
98
+ "6",
99
+ "7",
100
+ "8",
101
+ "9",
102
+ ".",
103
+ ",",
104
+ "!",
105
+ "?",
106
+ ";",
107
+ ":",
108
+ "\"",
109
+ "'",
110
+ "(",
111
+ ")",
112
+ "-",
113
+ "/",
114
+ "the",
115
+ "of",
116
+ "and",
117
+ "in",
118
+ "is",
119
+ "to",
120
+ "was",
121
+ "are",
122
+ "it",
123
+ "that",
124
+ "for",
125
+ "as",
126
+ "on",
127
+ "by",
128
+ "or",
129
+ "they",
130
+ "he",
131
+ "with",
132
+ "from",
133
+ "people",
134
+ "this",
135
+ "be",
136
+ "not",
137
+ "have",
138
+ "an",
139
+ "also",
140
+ "has",
141
+ "his",
142
+ "can",
143
+ "which",
144
+ "many",
145
+ "at",
146
+ "were",
147
+ "called",
148
+ "other",
149
+ "there",
150
+ "one",
151
+ "some",
152
+ "their",
153
+ "but",
154
+ "most",
155
+ "when",
156
+ "used",
157
+ "had",
158
+ "about",
159
+ "first",
160
+ "who",
161
+ "more",
162
+ "made",
163
+ "after",
164
+ "all",
165
+ "because",
166
+ "its",
167
+ "like",
168
+ "these",
169
+ "ing",
170
+ "tion",
171
+ "ion",
172
+ "er",
173
+ "ed",
174
+ "ly",
175
+ "al",
176
+ "es",
177
+ "en",
178
+ "st",
179
+ "le",
180
+ "re",
181
+ "ar",
182
+ "th",
183
+ "ch",
184
+ "sh",
185
+ "ou",
186
+ "ll",
187
+ "ment",
188
+ "ness"
189
+ ],
190
+ "word_tokens": [
191
+ "the",
192
+ "of",
193
+ "and",
194
+ "in",
195
+ "is",
196
+ "to",
197
+ "was",
198
+ "are",
199
+ "it",
200
+ "that",
201
+ "for",
202
+ "as",
203
+ "on",
204
+ "by",
205
+ "or",
206
+ "they",
207
+ "he",
208
+ "with",
209
+ "from",
210
+ "people",
211
+ "this",
212
+ "be",
213
+ "not",
214
+ "have",
215
+ "an",
216
+ "also",
217
+ "has",
218
+ "his",
219
+ "can",
220
+ "which",
221
+ "many",
222
+ "at",
223
+ "were",
224
+ "called",
225
+ "other",
226
+ "there",
227
+ "one",
228
+ "some",
229
+ "their",
230
+ "but",
231
+ "most",
232
+ "when",
233
+ "used",
234
+ "had",
235
+ "about",
236
+ "first",
237
+ "who",
238
+ "more",
239
+ "made",
240
+ "after",
241
+ "all",
242
+ "because",
243
+ "its",
244
+ "like",
245
+ "these"
246
+ ],
247
+ "subwords": [
248
+ "ing",
249
+ "tion",
250
+ "ion",
251
+ "er",
252
+ "ed",
253
+ "ly",
254
+ "al",
255
+ "es",
256
+ "en",
257
+ "st",
258
+ "le",
259
+ "re",
260
+ "ar",
261
+ "th",
262
+ "ch",
263
+ "sh",
264
+ "ou",
265
+ "ll",
266
+ "ment",
267
+ "ness"
268
+ ],
269
+ "punct": [
270
+ ".",
271
+ ",",
272
+ "!",
273
+ "?",
274
+ ";",
275
+ ":",
276
+ "\"",
277
+ "'",
278
+ "(",
279
+ ")",
280
+ "-",
281
+ "/"
282
+ ],
283
+ "digits": [
284
+ "0",
285
+ "1",
286
+ "2",
287
+ "3",
288
+ "4",
289
+ "5",
290
+ "6",
291
+ "7",
292
+ "8",
293
+ "9"
294
+ ]
295
+ }
296
+ }
v3_step_10000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f49b563dfd33dea437302c807f0c15c3f4b80604f569e29dcd74d5523049ed4b
3
+ size 20130527
v3_step_15000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:79be5f2d7e57423448ce0849bd92667fd844aff0597b458d2a7565bb5b707b6f
3
+ size 20130527
v3_step_20000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5c6ae04284d553cba433e3963edafa5a4bdae7841f3c10ecde87f1353b3c03be
3
+ size 20130527
v3_step_5000.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:141af671c5c2260ea8dba73ee5dc841e7107d10a3e8fbb0701aa7d292f589dab
3
+ size 20130449
v3_tokenizer.json ADDED
@@ -0,0 +1,238 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": "tiny-qwen-v2-boundary-128",
3
+ "vocab": [
4
+ "<pad>",
5
+ "<unk>",
6
+ "<bos>",
7
+ "<eos>",
8
+ " ",
9
+ "a",
10
+ "b",
11
+ "c",
12
+ "d",
13
+ "e",
14
+ "f",
15
+ "g",
16
+ "h",
17
+ "i",
18
+ "j",
19
+ "k",
20
+ "l",
21
+ "m",
22
+ "n",
23
+ "o",
24
+ "p",
25
+ "q",
26
+ "r",
27
+ "s",
28
+ "t",
29
+ "u",
30
+ "v",
31
+ "w",
32
+ "x",
33
+ "y",
34
+ "z",
35
+ "0",
36
+ "1",
37
+ "2",
38
+ "3",
39
+ "4",
40
+ "5",
41
+ "6",
42
+ "7",
43
+ "8",
44
+ "9",
45
+ ".",
46
+ ",",
47
+ "!",
48
+ "?",
49
+ ";",
50
+ ":",
51
+ "\"",
52
+ "'",
53
+ "(",
54
+ ")",
55
+ "-",
56
+ "/",
57
+ "the",
58
+ "of",
59
+ "and",
60
+ "in",
61
+ "is",
62
+ "to",
63
+ "was",
64
+ "are",
65
+ "it",
66
+ "that",
67
+ "for",
68
+ "as",
69
+ "on",
70
+ "by",
71
+ "or",
72
+ "they",
73
+ "he",
74
+ "with",
75
+ "from",
76
+ "people",
77
+ "this",
78
+ "be",
79
+ "not",
80
+ "have",
81
+ "an",
82
+ "also",
83
+ "has",
84
+ "his",
85
+ "can",
86
+ "which",
87
+ "many",
88
+ "at",
89
+ "were",
90
+ "called",
91
+ "other",
92
+ "there",
93
+ "one",
94
+ "some",
95
+ "their",
96
+ "but",
97
+ "most",
98
+ "when",
99
+ "used",
100
+ "had",
101
+ "about",
102
+ "first",
103
+ "who",
104
+ "more",
105
+ "made",
106
+ "after",
107
+ "all",
108
+ "because",
109
+ "its",
110
+ "like",
111
+ "these",
112
+ "ing",
113
+ "tion",
114
+ "ion",
115
+ "er",
116
+ "ed",
117
+ "ly",
118
+ "al",
119
+ "es",
120
+ "en",
121
+ "st",
122
+ "le",
123
+ "re",
124
+ "ar",
125
+ "th",
126
+ "ch",
127
+ "sh",
128
+ "ou",
129
+ "ll",
130
+ "ment",
131
+ "ness"
132
+ ],
133
+ "word_tokens": [
134
+ "the",
135
+ "of",
136
+ "and",
137
+ "in",
138
+ "is",
139
+ "to",
140
+ "was",
141
+ "are",
142
+ "it",
143
+ "that",
144
+ "for",
145
+ "as",
146
+ "on",
147
+ "by",
148
+ "or",
149
+ "they",
150
+ "he",
151
+ "with",
152
+ "from",
153
+ "people",
154
+ "this",
155
+ "be",
156
+ "not",
157
+ "have",
158
+ "an",
159
+ "also",
160
+ "has",
161
+ "his",
162
+ "can",
163
+ "which",
164
+ "many",
165
+ "at",
166
+ "were",
167
+ "called",
168
+ "other",
169
+ "there",
170
+ "one",
171
+ "some",
172
+ "their",
173
+ "but",
174
+ "most",
175
+ "when",
176
+ "used",
177
+ "had",
178
+ "about",
179
+ "first",
180
+ "who",
181
+ "more",
182
+ "made",
183
+ "after",
184
+ "all",
185
+ "because",
186
+ "its",
187
+ "like",
188
+ "these"
189
+ ],
190
+ "subwords": [
191
+ "ing",
192
+ "tion",
193
+ "ion",
194
+ "er",
195
+ "ed",
196
+ "ly",
197
+ "al",
198
+ "es",
199
+ "en",
200
+ "st",
201
+ "le",
202
+ "re",
203
+ "ar",
204
+ "th",
205
+ "ch",
206
+ "sh",
207
+ "ou",
208
+ "ll",
209
+ "ment",
210
+ "ness"
211
+ ],
212
+ "punct": [
213
+ ".",
214
+ ",",
215
+ "!",
216
+ "?",
217
+ ";",
218
+ ":",
219
+ "\"",
220
+ "'",
221
+ "(",
222
+ ")",
223
+ "-",
224
+ "/"
225
+ ],
226
+ "digits": [
227
+ "0",
228
+ "1",
229
+ "2",
230
+ "3",
231
+ "4",
232
+ "5",
233
+ "6",
234
+ "7",
235
+ "8",
236
+ "9"
237
+ ]
238
+ }