nlparabic commited on
Commit
f983c76
1 Parent(s): 1859732

Training in progress, epoch 1

Browse files
added_tokens.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "<sep>": 64003,
3
+ "<|bos|>": 64000,
4
+ "<|unk|>": 64001,
5
+ "[PAD]": 64002
6
+ }
config.json ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "aubmindlab/aragpt2-large",
3
+ "activation_function": "gelu_new",
4
+ "architectures": [
5
+ "AraGPT2LMHeadModel"
6
+ ],
7
+ "attention_probs_dropout_prob": 0.1,
8
+ "attn_pdrop": 0.1,
9
+ "auto_map": {
10
+ "AutoConfig": "aubmindlab/aragpt2-large--configuration_aragpt2.AraGPT2Config",
11
+ "AutoModel": "aubmindlab/aragpt2-large--modeling_aragpt2.AraGPT2Model",
12
+ "AutoModelForCausalLM": "aubmindlab/aragpt2-large--modeling_aragpt2.AraGPT2LMHeadModel"
13
+ },
14
+ "bos_token_id": 0,
15
+ "embd_pdrop": 0.1,
16
+ "eos_token_id": 0,
17
+ "hidden_act": "gelu",
18
+ "hidden_dropout_prob": 0.1,
19
+ "initializer_range": 0.014142135623731,
20
+ "intermediate_size": 5120,
21
+ "layer_norm_epsilon": 1e-05,
22
+ "model_type": "aragpt2",
23
+ "n_ctx": 1024,
24
+ "n_embd": 1280,
25
+ "n_head": 20,
26
+ "n_inner": null,
27
+ "n_layer": 36,
28
+ "n_positions": 1024,
29
+ "reorder_and_upcast_attn": false,
30
+ "resid_pdrop": 0.1,
31
+ "scale_attn_by_inverse_layer_idx": false,
32
+ "scale_attn_weights": true,
33
+ "summary_activation": null,
34
+ "summary_first_dropout": 0.1,
35
+ "summary_proj_to_labels": true,
36
+ "summary_type": "cls_index",
37
+ "summary_use_proj": true,
38
+ "task_specific_params": {
39
+ "text-generation": {
40
+ "do_sample": true,
41
+ "max_length": 50,
42
+ "no_repeat_ngram_size": 3,
43
+ "num_beams": 5,
44
+ "repetition_penalty": 3.0,
45
+ "top_p": 0.95
46
+ }
47
+ },
48
+ "tokenizer_class": "GPT2Tokenizer",
49
+ "torch_dtype": "float32",
50
+ "transformers_version": "4.45.0.dev0",
51
+ "use_cache": true,
52
+ "vocab_size": 64004
53
+ }
egy_training_log.txt ADDED
@@ -0,0 +1,700 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
2
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
3
+ _n_gpu=1,
4
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
5
+ adafactor=False,
6
+ adam_beta1=0.9,
7
+ adam_beta2=0.999,
8
+ adam_epsilon=1e-08,
9
+ auto_find_batch_size=False,
10
+ batch_eval_metrics=False,
11
+ bf16=False,
12
+ bf16_full_eval=False,
13
+ data_seed=None,
14
+ dataloader_drop_last=False,
15
+ dataloader_num_workers=0,
16
+ dataloader_persistent_workers=False,
17
+ dataloader_pin_memory=True,
18
+ dataloader_prefetch_factor=None,
19
+ ddp_backend=None,
20
+ ddp_broadcast_buffers=None,
21
+ ddp_bucket_cap_mb=None,
22
+ ddp_find_unused_parameters=None,
23
+ ddp_timeout=1800,
24
+ debug=[],
25
+ deepspeed=None,
26
+ disable_tqdm=False,
27
+ dispatch_batches=None,
28
+ do_eval=True,
29
+ do_predict=False,
30
+ do_train=True,
31
+ eval_accumulation_steps=None,
32
+ eval_delay=0,
33
+ eval_do_concat_batches=True,
34
+ eval_on_start=False,
35
+ eval_steps=None,
36
+ eval_strategy=IntervalStrategy.EPOCH,
37
+ eval_use_gather_object=False,
38
+ evaluation_strategy=epoch,
39
+ fp16=False,
40
+ fp16_backend=auto,
41
+ fp16_full_eval=False,
42
+ fp16_opt_level=O1,
43
+ fsdp=[],
44
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
45
+ fsdp_min_num_params=0,
46
+ fsdp_transformer_layer_cls_to_wrap=None,
47
+ full_determinism=False,
48
+ gradient_accumulation_steps=1,
49
+ gradient_checkpointing=False,
50
+ gradient_checkpointing_kwargs=None,
51
+ greater_is_better=False,
52
+ group_by_length=False,
53
+ half_precision_backend=auto,
54
+ hub_always_push=False,
55
+ hub_model_id=None,
56
+ hub_private_repo=False,
57
+ hub_strategy=HubStrategy.EVERY_SAVE,
58
+ hub_token=<HUB_TOKEN>,
59
+ ignore_data_skip=False,
60
+ include_inputs_for_metrics=False,
61
+ include_num_input_tokens_seen=False,
62
+ include_tokens_per_second=False,
63
+ jit_mode_eval=False,
64
+ label_names=None,
65
+ label_smoothing_factor=0.0,
66
+ learning_rate=5e-05,
67
+ length_column_name=length,
68
+ load_best_model_at_end=True,
69
+ local_rank=0,
70
+ log_level=passive,
71
+ log_level_replica=warning,
72
+ log_on_each_node=True,
73
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large/runs/Sep08_10-04-47_lmgpu-node-07,
74
+ logging_first_step=False,
75
+ logging_nan_inf_filter=True,
76
+ logging_steps=500,
77
+ logging_strategy=IntervalStrategy.EPOCH,
78
+ lr_scheduler_kwargs={},
79
+ lr_scheduler_type=SchedulerType.LINEAR,
80
+ max_grad_norm=1.0,
81
+ max_steps=-1,
82
+ metric_for_best_model=loss,
83
+ mp_parameters=,
84
+ neftune_noise_alpha=None,
85
+ no_cuda=False,
86
+ num_train_epochs=20.0,
87
+ optim=OptimizerNames.ADAMW_TORCH,
88
+ optim_args=None,
89
+ optim_target_modules=None,
90
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
91
+ overwrite_output_dir=False,
92
+ past_index=-1,
93
+ per_device_eval_batch_size=8,
94
+ per_device_train_batch_size=8,
95
+ prediction_loss_only=False,
96
+ push_to_hub=True,
97
+ push_to_hub_model_id=None,
98
+ push_to_hub_organization=None,
99
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
100
+ ray_scope=last,
101
+ remove_unused_columns=True,
102
+ report_to=[],
103
+ restore_callback_states_from_checkpoint=False,
104
+ resume_from_checkpoint=None,
105
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
106
+ save_on_each_node=False,
107
+ save_only_model=False,
108
+ save_safetensors=True,
109
+ save_steps=500,
110
+ save_strategy=IntervalStrategy.EPOCH,
111
+ save_total_limit=None,
112
+ seed=42,
113
+ skip_memory_metrics=True,
114
+ split_batches=None,
115
+ tf32=None,
116
+ torch_compile=False,
117
+ torch_compile_backend=None,
118
+ torch_compile_mode=None,
119
+ torch_empty_cache_steps=None,
120
+ torchdynamo=None,
121
+ tpu_metrics_debug=False,
122
+ tpu_num_cores=None,
123
+ use_cpu=False,
124
+ use_ipex=False,
125
+ use_legacy_prediction_loop=False,
126
+ use_mps_device=False,
127
+ warmup_ratio=0.0,
128
+ warmup_steps=500,
129
+ weight_decay=0.0,
130
+ )
131
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
132
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
133
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
134
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
135
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
136
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
137
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
138
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
139
+ _n_gpu=1,
140
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
141
+ adafactor=False,
142
+ adam_beta1=0.9,
143
+ adam_beta2=0.999,
144
+ adam_epsilon=1e-08,
145
+ auto_find_batch_size=False,
146
+ batch_eval_metrics=False,
147
+ bf16=False,
148
+ bf16_full_eval=False,
149
+ data_seed=None,
150
+ dataloader_drop_last=False,
151
+ dataloader_num_workers=0,
152
+ dataloader_persistent_workers=False,
153
+ dataloader_pin_memory=True,
154
+ dataloader_prefetch_factor=None,
155
+ ddp_backend=None,
156
+ ddp_broadcast_buffers=None,
157
+ ddp_bucket_cap_mb=None,
158
+ ddp_find_unused_parameters=None,
159
+ ddp_timeout=1800,
160
+ debug=[],
161
+ deepspeed=None,
162
+ disable_tqdm=False,
163
+ dispatch_batches=None,
164
+ do_eval=True,
165
+ do_predict=False,
166
+ do_train=True,
167
+ eval_accumulation_steps=None,
168
+ eval_delay=0,
169
+ eval_do_concat_batches=True,
170
+ eval_on_start=False,
171
+ eval_steps=None,
172
+ eval_strategy=IntervalStrategy.EPOCH,
173
+ eval_use_gather_object=False,
174
+ evaluation_strategy=epoch,
175
+ fp16=False,
176
+ fp16_backend=auto,
177
+ fp16_full_eval=False,
178
+ fp16_opt_level=O1,
179
+ fsdp=[],
180
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
181
+ fsdp_min_num_params=0,
182
+ fsdp_transformer_layer_cls_to_wrap=None,
183
+ full_determinism=False,
184
+ gradient_accumulation_steps=1,
185
+ gradient_checkpointing=False,
186
+ gradient_checkpointing_kwargs=None,
187
+ greater_is_better=False,
188
+ group_by_length=False,
189
+ half_precision_backend=auto,
190
+ hub_always_push=False,
191
+ hub_model_id=None,
192
+ hub_private_repo=False,
193
+ hub_strategy=HubStrategy.EVERY_SAVE,
194
+ hub_token=<HUB_TOKEN>,
195
+ ignore_data_skip=False,
196
+ include_inputs_for_metrics=False,
197
+ include_num_input_tokens_seen=False,
198
+ include_tokens_per_second=False,
199
+ jit_mode_eval=False,
200
+ label_names=None,
201
+ label_smoothing_factor=0.0,
202
+ learning_rate=5e-05,
203
+ length_column_name=length,
204
+ load_best_model_at_end=True,
205
+ local_rank=0,
206
+ log_level=passive,
207
+ log_level_replica=warning,
208
+ log_on_each_node=True,
209
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large/runs/Sep08_10-08-45_lmgpu-node-07,
210
+ logging_first_step=False,
211
+ logging_nan_inf_filter=True,
212
+ logging_steps=500,
213
+ logging_strategy=IntervalStrategy.EPOCH,
214
+ lr_scheduler_kwargs={},
215
+ lr_scheduler_type=SchedulerType.LINEAR,
216
+ max_grad_norm=1.0,
217
+ max_steps=-1,
218
+ metric_for_best_model=loss,
219
+ mp_parameters=,
220
+ neftune_noise_alpha=None,
221
+ no_cuda=False,
222
+ num_train_epochs=20.0,
223
+ optim=OptimizerNames.ADAMW_TORCH,
224
+ optim_args=None,
225
+ optim_target_modules=None,
226
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
227
+ overwrite_output_dir=False,
228
+ past_index=-1,
229
+ per_device_eval_batch_size=8,
230
+ per_device_train_batch_size=8,
231
+ prediction_loss_only=False,
232
+ push_to_hub=True,
233
+ push_to_hub_model_id=None,
234
+ push_to_hub_organization=None,
235
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
236
+ ray_scope=last,
237
+ remove_unused_columns=True,
238
+ report_to=[],
239
+ restore_callback_states_from_checkpoint=False,
240
+ resume_from_checkpoint=None,
241
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
242
+ save_on_each_node=False,
243
+ save_only_model=False,
244
+ save_safetensors=True,
245
+ save_steps=500,
246
+ save_strategy=IntervalStrategy.EPOCH,
247
+ save_total_limit=None,
248
+ seed=42,
249
+ skip_memory_metrics=True,
250
+ split_batches=None,
251
+ tf32=None,
252
+ torch_compile=False,
253
+ torch_compile_backend=None,
254
+ torch_compile_mode=None,
255
+ torch_empty_cache_steps=None,
256
+ torchdynamo=None,
257
+ tpu_metrics_debug=False,
258
+ tpu_num_cores=None,
259
+ use_cpu=False,
260
+ use_ipex=False,
261
+ use_legacy_prediction_loop=False,
262
+ use_mps_device=False,
263
+ warmup_ratio=0.0,
264
+ warmup_steps=500,
265
+ weight_decay=0.0,
266
+ )
267
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
268
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
269
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
270
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
271
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
272
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
273
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
274
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
275
+ _n_gpu=1,
276
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
277
+ adafactor=False,
278
+ adam_beta1=0.9,
279
+ adam_beta2=0.999,
280
+ adam_epsilon=1e-08,
281
+ auto_find_batch_size=False,
282
+ batch_eval_metrics=False,
283
+ bf16=False,
284
+ bf16_full_eval=False,
285
+ data_seed=None,
286
+ dataloader_drop_last=False,
287
+ dataloader_num_workers=0,
288
+ dataloader_persistent_workers=False,
289
+ dataloader_pin_memory=True,
290
+ dataloader_prefetch_factor=None,
291
+ ddp_backend=None,
292
+ ddp_broadcast_buffers=None,
293
+ ddp_bucket_cap_mb=None,
294
+ ddp_find_unused_parameters=None,
295
+ ddp_timeout=1800,
296
+ debug=[],
297
+ deepspeed=None,
298
+ disable_tqdm=False,
299
+ dispatch_batches=None,
300
+ do_eval=True,
301
+ do_predict=False,
302
+ do_train=True,
303
+ eval_accumulation_steps=None,
304
+ eval_delay=0,
305
+ eval_do_concat_batches=True,
306
+ eval_on_start=False,
307
+ eval_steps=None,
308
+ eval_strategy=IntervalStrategy.EPOCH,
309
+ eval_use_gather_object=False,
310
+ evaluation_strategy=epoch,
311
+ fp16=False,
312
+ fp16_backend=auto,
313
+ fp16_full_eval=False,
314
+ fp16_opt_level=O1,
315
+ fsdp=[],
316
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
317
+ fsdp_min_num_params=0,
318
+ fsdp_transformer_layer_cls_to_wrap=None,
319
+ full_determinism=False,
320
+ gradient_accumulation_steps=1,
321
+ gradient_checkpointing=False,
322
+ gradient_checkpointing_kwargs=None,
323
+ greater_is_better=False,
324
+ group_by_length=False,
325
+ half_precision_backend=auto,
326
+ hub_always_push=False,
327
+ hub_model_id=None,
328
+ hub_private_repo=False,
329
+ hub_strategy=HubStrategy.EVERY_SAVE,
330
+ hub_token=<HUB_TOKEN>,
331
+ ignore_data_skip=False,
332
+ include_inputs_for_metrics=False,
333
+ include_num_input_tokens_seen=False,
334
+ include_tokens_per_second=False,
335
+ jit_mode_eval=False,
336
+ label_names=None,
337
+ label_smoothing_factor=0.0,
338
+ learning_rate=5e-05,
339
+ length_column_name=length,
340
+ load_best_model_at_end=True,
341
+ local_rank=0,
342
+ log_level=passive,
343
+ log_level_replica=warning,
344
+ log_on_each_node=True,
345
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large/runs/Sep08_10-12-27_lmgpu-node-07,
346
+ logging_first_step=False,
347
+ logging_nan_inf_filter=True,
348
+ logging_steps=500,
349
+ logging_strategy=IntervalStrategy.EPOCH,
350
+ lr_scheduler_kwargs={},
351
+ lr_scheduler_type=SchedulerType.LINEAR,
352
+ max_grad_norm=1.0,
353
+ max_steps=-1,
354
+ metric_for_best_model=loss,
355
+ mp_parameters=,
356
+ neftune_noise_alpha=None,
357
+ no_cuda=False,
358
+ num_train_epochs=20.0,
359
+ optim=OptimizerNames.ADAMW_TORCH,
360
+ optim_args=None,
361
+ optim_target_modules=None,
362
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
363
+ overwrite_output_dir=False,
364
+ past_index=-1,
365
+ per_device_eval_batch_size=8,
366
+ per_device_train_batch_size=8,
367
+ prediction_loss_only=False,
368
+ push_to_hub=True,
369
+ push_to_hub_model_id=None,
370
+ push_to_hub_organization=None,
371
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
372
+ ray_scope=last,
373
+ remove_unused_columns=True,
374
+ report_to=[],
375
+ restore_callback_states_from_checkpoint=False,
376
+ resume_from_checkpoint=None,
377
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
378
+ save_on_each_node=False,
379
+ save_only_model=False,
380
+ save_safetensors=True,
381
+ save_steps=500,
382
+ save_strategy=IntervalStrategy.EPOCH,
383
+ save_total_limit=None,
384
+ seed=42,
385
+ skip_memory_metrics=True,
386
+ split_batches=None,
387
+ tf32=None,
388
+ torch_compile=False,
389
+ torch_compile_backend=None,
390
+ torch_compile_mode=None,
391
+ torch_empty_cache_steps=None,
392
+ torchdynamo=None,
393
+ tpu_metrics_debug=False,
394
+ tpu_num_cores=None,
395
+ use_cpu=False,
396
+ use_ipex=False,
397
+ use_legacy_prediction_loop=False,
398
+ use_mps_device=False,
399
+ warmup_ratio=0.0,
400
+ warmup_steps=500,
401
+ weight_decay=0.0,
402
+ )
403
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
404
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
405
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
406
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
407
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
408
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
409
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-b54442fb13b02a61.arrow
410
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-6c1ce86d0282180c.arrow
411
+ WARNING:__main__:The tokenizer picked seems to have a very large `model_max_length` (1000000000000000019884624838656). Using block_size=1024 instead. You can change that default value by passing --block_size xxx.
412
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-c2dc36fb1e81081d.arrow
413
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-1909b7a07e479059.arrow
414
+ WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
415
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
416
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
417
+ _n_gpu=1,
418
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
419
+ adafactor=False,
420
+ adam_beta1=0.9,
421
+ adam_beta2=0.999,
422
+ adam_epsilon=1e-08,
423
+ auto_find_batch_size=False,
424
+ batch_eval_metrics=False,
425
+ bf16=False,
426
+ bf16_full_eval=False,
427
+ data_seed=None,
428
+ dataloader_drop_last=False,
429
+ dataloader_num_workers=0,
430
+ dataloader_persistent_workers=False,
431
+ dataloader_pin_memory=True,
432
+ dataloader_prefetch_factor=None,
433
+ ddp_backend=None,
434
+ ddp_broadcast_buffers=None,
435
+ ddp_bucket_cap_mb=None,
436
+ ddp_find_unused_parameters=None,
437
+ ddp_timeout=1800,
438
+ debug=[],
439
+ deepspeed=None,
440
+ disable_tqdm=False,
441
+ dispatch_batches=None,
442
+ do_eval=True,
443
+ do_predict=False,
444
+ do_train=True,
445
+ eval_accumulation_steps=None,
446
+ eval_delay=0,
447
+ eval_do_concat_batches=True,
448
+ eval_on_start=False,
449
+ eval_steps=None,
450
+ eval_strategy=IntervalStrategy.EPOCH,
451
+ eval_use_gather_object=False,
452
+ evaluation_strategy=epoch,
453
+ fp16=False,
454
+ fp16_backend=auto,
455
+ fp16_full_eval=False,
456
+ fp16_opt_level=O1,
457
+ fsdp=[],
458
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
459
+ fsdp_min_num_params=0,
460
+ fsdp_transformer_layer_cls_to_wrap=None,
461
+ full_determinism=False,
462
+ gradient_accumulation_steps=1,
463
+ gradient_checkpointing=False,
464
+ gradient_checkpointing_kwargs=None,
465
+ greater_is_better=False,
466
+ group_by_length=False,
467
+ half_precision_backend=auto,
468
+ hub_always_push=False,
469
+ hub_model_id=None,
470
+ hub_private_repo=False,
471
+ hub_strategy=HubStrategy.EVERY_SAVE,
472
+ hub_token=<HUB_TOKEN>,
473
+ ignore_data_skip=False,
474
+ include_inputs_for_metrics=False,
475
+ include_num_input_tokens_seen=False,
476
+ include_tokens_per_second=False,
477
+ jit_mode_eval=False,
478
+ label_names=None,
479
+ label_smoothing_factor=0.0,
480
+ learning_rate=5e-05,
481
+ length_column_name=length,
482
+ load_best_model_at_end=True,
483
+ local_rank=0,
484
+ log_level=passive,
485
+ log_level_replica=warning,
486
+ log_on_each_node=True,
487
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large/runs/Sep08_10-16-23_lmgpu-node-07,
488
+ logging_first_step=False,
489
+ logging_nan_inf_filter=True,
490
+ logging_steps=500,
491
+ logging_strategy=IntervalStrategy.EPOCH,
492
+ lr_scheduler_kwargs={},
493
+ lr_scheduler_type=SchedulerType.LINEAR,
494
+ max_grad_norm=1.0,
495
+ max_steps=-1,
496
+ metric_for_best_model=loss,
497
+ mp_parameters=,
498
+ neftune_noise_alpha=None,
499
+ no_cuda=False,
500
+ num_train_epochs=20.0,
501
+ optim=OptimizerNames.ADAMW_TORCH,
502
+ optim_args=None,
503
+ optim_target_modules=None,
504
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
505
+ overwrite_output_dir=False,
506
+ past_index=-1,
507
+ per_device_eval_batch_size=4,
508
+ per_device_train_batch_size=4,
509
+ prediction_loss_only=False,
510
+ push_to_hub=True,
511
+ push_to_hub_model_id=None,
512
+ push_to_hub_organization=None,
513
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
514
+ ray_scope=last,
515
+ remove_unused_columns=True,
516
+ report_to=[],
517
+ restore_callback_states_from_checkpoint=False,
518
+ resume_from_checkpoint=None,
519
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
520
+ save_on_each_node=False,
521
+ save_only_model=False,
522
+ save_safetensors=True,
523
+ save_steps=500,
524
+ save_strategy=IntervalStrategy.EPOCH,
525
+ save_total_limit=None,
526
+ seed=42,
527
+ skip_memory_metrics=True,
528
+ split_batches=None,
529
+ tf32=None,
530
+ torch_compile=False,
531
+ torch_compile_backend=None,
532
+ torch_compile_mode=None,
533
+ torch_empty_cache_steps=None,
534
+ torchdynamo=None,
535
+ tpu_metrics_debug=False,
536
+ tpu_num_cores=None,
537
+ use_cpu=False,
538
+ use_ipex=False,
539
+ use_legacy_prediction_loop=False,
540
+ use_mps_device=False,
541
+ warmup_ratio=0.0,
542
+ warmup_steps=500,
543
+ weight_decay=0.0,
544
+ )
545
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
546
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
547
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
548
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
549
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
550
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
551
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-b54442fb13b02a61.arrow
552
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-b1df2d2ea350d912.arrow
553
+ WARNING:__main__:The tokenizer picked seems to have a very large `model_max_length` (1000000000000000019884624838656). Using block_size=1024 instead. You can change that default value by passing --block_size xxx.
554
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-c2dc36fb1e81081d.arrow
555
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-a86a326980c048c0.arrow
556
+ WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
557
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
558
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
559
+ _n_gpu=1,
560
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
561
+ adafactor=False,
562
+ adam_beta1=0.9,
563
+ adam_beta2=0.999,
564
+ adam_epsilon=1e-08,
565
+ auto_find_batch_size=False,
566
+ batch_eval_metrics=False,
567
+ bf16=False,
568
+ bf16_full_eval=False,
569
+ data_seed=None,
570
+ dataloader_drop_last=False,
571
+ dataloader_num_workers=0,
572
+ dataloader_persistent_workers=False,
573
+ dataloader_pin_memory=True,
574
+ dataloader_prefetch_factor=None,
575
+ ddp_backend=None,
576
+ ddp_broadcast_buffers=None,
577
+ ddp_bucket_cap_mb=None,
578
+ ddp_find_unused_parameters=None,
579
+ ddp_timeout=1800,
580
+ debug=[],
581
+ deepspeed=None,
582
+ disable_tqdm=False,
583
+ dispatch_batches=None,
584
+ do_eval=True,
585
+ do_predict=False,
586
+ do_train=True,
587
+ eval_accumulation_steps=None,
588
+ eval_delay=0,
589
+ eval_do_concat_batches=True,
590
+ eval_on_start=False,
591
+ eval_steps=None,
592
+ eval_strategy=IntervalStrategy.EPOCH,
593
+ eval_use_gather_object=False,
594
+ evaluation_strategy=epoch,
595
+ fp16=False,
596
+ fp16_backend=auto,
597
+ fp16_full_eval=False,
598
+ fp16_opt_level=O1,
599
+ fsdp=[],
600
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
601
+ fsdp_min_num_params=0,
602
+ fsdp_transformer_layer_cls_to_wrap=None,
603
+ full_determinism=False,
604
+ gradient_accumulation_steps=1,
605
+ gradient_checkpointing=False,
606
+ gradient_checkpointing_kwargs=None,
607
+ greater_is_better=False,
608
+ group_by_length=False,
609
+ half_precision_backend=auto,
610
+ hub_always_push=False,
611
+ hub_model_id=None,
612
+ hub_private_repo=False,
613
+ hub_strategy=HubStrategy.EVERY_SAVE,
614
+ hub_token=<HUB_TOKEN>,
615
+ ignore_data_skip=False,
616
+ include_inputs_for_metrics=False,
617
+ include_num_input_tokens_seen=False,
618
+ include_tokens_per_second=False,
619
+ jit_mode_eval=False,
620
+ label_names=None,
621
+ label_smoothing_factor=0.0,
622
+ learning_rate=5e-05,
623
+ length_column_name=length,
624
+ load_best_model_at_end=True,
625
+ local_rank=0,
626
+ log_level=passive,
627
+ log_level_replica=warning,
628
+ log_on_each_node=True,
629
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large/runs/Sep08_11-01-05_lmgpu-node-07,
630
+ logging_first_step=False,
631
+ logging_nan_inf_filter=True,
632
+ logging_steps=500,
633
+ logging_strategy=IntervalStrategy.EPOCH,
634
+ lr_scheduler_kwargs={},
635
+ lr_scheduler_type=SchedulerType.LINEAR,
636
+ max_grad_norm=1.0,
637
+ max_steps=-1,
638
+ metric_for_best_model=loss,
639
+ mp_parameters=,
640
+ neftune_noise_alpha=None,
641
+ no_cuda=False,
642
+ num_train_epochs=20.0,
643
+ optim=OptimizerNames.ADAMW_TORCH,
644
+ optim_args=None,
645
+ optim_target_modules=None,
646
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
647
+ overwrite_output_dir=False,
648
+ past_index=-1,
649
+ per_device_eval_batch_size=4,
650
+ per_device_train_batch_size=4,
651
+ prediction_loss_only=False,
652
+ push_to_hub=True,
653
+ push_to_hub_model_id=None,
654
+ push_to_hub_organization=None,
655
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
656
+ ray_scope=last,
657
+ remove_unused_columns=True,
658
+ report_to=[],
659
+ restore_callback_states_from_checkpoint=False,
660
+ resume_from_checkpoint=None,
661
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-large,
662
+ save_on_each_node=False,
663
+ save_only_model=False,
664
+ save_safetensors=True,
665
+ save_steps=500,
666
+ save_strategy=IntervalStrategy.EPOCH,
667
+ save_total_limit=None,
668
+ seed=42,
669
+ skip_memory_metrics=True,
670
+ split_batches=None,
671
+ tf32=None,
672
+ torch_compile=False,
673
+ torch_compile_backend=None,
674
+ torch_compile_mode=None,
675
+ torch_empty_cache_steps=None,
676
+ torchdynamo=None,
677
+ tpu_metrics_debug=False,
678
+ tpu_num_cores=None,
679
+ use_cpu=False,
680
+ use_ipex=False,
681
+ use_legacy_prediction_loop=False,
682
+ use_mps_device=False,
683
+ warmup_ratio=0.0,
684
+ warmup_steps=500,
685
+ weight_decay=0.0,
686
+ )
687
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
688
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
689
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
690
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
691
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
692
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
693
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-b54442fb13b02a61.arrow
694
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-b1df2d2ea350d912.arrow
695
+ WARNING:__main__:The tokenizer picked seems to have a very large `model_max_length` (1000000000000000019884624838656). Using block_size=1024 instead. You can change that default value by passing --block_size xxx.
696
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-c2dc36fb1e81081d.arrow
697
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-a86a326980c048c0.arrow
698
+ WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
699
+ INFO:root:Epoch 1.0: Train Loss = None, Eval Loss = None
700
+ INFO:absl:Using default tokenizer.
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:55a53c4ea4b3c235093546256ff77e07a847f109fc6c0fb020e419e60c728c1e
3
+ size 3166550552
special_tokens_map.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ {
4
+ "content": "<sep>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false
9
+ }
10
+ ],
11
+ "bos_token": {
12
+ "content": "<|bos|>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false
17
+ },
18
+ "eos_token": {
19
+ "content": "<|endoftext|>",
20
+ "lstrip": false,
21
+ "normalized": false,
22
+ "rstrip": false,
23
+ "single_word": false
24
+ },
25
+ "pad_token": {
26
+ "content": "[PAD]",
27
+ "lstrip": false,
28
+ "normalized": false,
29
+ "rstrip": false,
30
+ "single_word": false
31
+ },
32
+ "unk_token": {
33
+ "content": "<|unk|>",
34
+ "lstrip": false,
35
+ "normalized": false,
36
+ "rstrip": false,
37
+ "single_word": false
38
+ }
39
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "0": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "1": {
13
+ "content": "<s>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "2": {
21
+ "content": "<pad>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "3": {
29
+ "content": "</s>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "64000": {
37
+ "content": "<|bos|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "64001": {
45
+ "content": "<|unk|>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "64002": {
53
+ "content": "[PAD]",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "64003": {
61
+ "content": "<sep>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ }
68
+ },
69
+ "additional_special_tokens": [
70
+ "<sep>"
71
+ ],
72
+ "bos_token": "<|bos|>",
73
+ "clean_up_tokenization_spaces": true,
74
+ "eos_token": "<|endoftext|>",
75
+ "model_max_length": 1000000000000000019884624838656,
76
+ "pad_token": "[PAD]",
77
+ "tokenizer_class": "GPT2Tokenizer",
78
+ "unk_token": "<|unk|>"
79
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:80413d95dbf08e485be3bbb75758aba391e927d9025050d10e18a3378ccab644
3
+ size 5240
vocab.json ADDED
The diff for this file is too large to render. See raw diff