nlparabic commited on
Commit
18fb08d
·
verified ·
1 Parent(s): fecca88

Training in progress, epoch 1

Browse files
added_tokens.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "<sep>": 64003,
3
+ "<|bos|>": 64000,
4
+ "<|unk|>": 64001,
5
+ "[PAD]": 64002
6
+ }
config.json ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "aubmindlab/aragpt2-base",
3
+ "activation_function": "gelu_new",
4
+ "architectures": [
5
+ "GPT2LMHeadModel"
6
+ ],
7
+ "attn_pdrop": 0.1,
8
+ "bos_token_id": 0,
9
+ "embd_pdrop": 0.1,
10
+ "eos_token_id": 0,
11
+ "gradient_checkpointing": false,
12
+ "initializer_range": 0.02,
13
+ "layer_norm_epsilon": 1e-05,
14
+ "model_type": "gpt2",
15
+ "n_ctx": 1024,
16
+ "n_embd": 768,
17
+ "n_head": 12,
18
+ "n_inner": null,
19
+ "n_layer": 12,
20
+ "n_positions": 1024,
21
+ "reorder_and_upcast_attn": false,
22
+ "resid_pdrop": 0.1,
23
+ "scale_attn_by_inverse_layer_idx": false,
24
+ "scale_attn_weights": true,
25
+ "summary_activation": null,
26
+ "summary_first_dropout": 0.1,
27
+ "summary_proj_to_labels": true,
28
+ "summary_type": "cls_index",
29
+ "summary_use_proj": true,
30
+ "task_specific_params": {
31
+ "text-generation": {
32
+ "do_sample": true,
33
+ "max_length": 50,
34
+ "no_repeat_ngram_size": 3,
35
+ "num_beams": 5,
36
+ "repetition_penalty": 3.0,
37
+ "top_p": 0.95
38
+ }
39
+ },
40
+ "torch_dtype": "float32",
41
+ "transformers_version": "4.45.0.dev0",
42
+ "use_cache": true,
43
+ "vocab_size": 64004
44
+ }
egy_training_log.txt ADDED
@@ -0,0 +1,1918 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
2
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
3
+ _n_gpu=1,
4
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
5
+ adafactor=False,
6
+ adam_beta1=0.9,
7
+ adam_beta2=0.999,
8
+ adam_epsilon=1e-08,
9
+ auto_find_batch_size=False,
10
+ batch_eval_metrics=False,
11
+ bf16=False,
12
+ bf16_full_eval=False,
13
+ data_seed=None,
14
+ dataloader_drop_last=False,
15
+ dataloader_num_workers=0,
16
+ dataloader_persistent_workers=False,
17
+ dataloader_pin_memory=True,
18
+ dataloader_prefetch_factor=None,
19
+ ddp_backend=None,
20
+ ddp_broadcast_buffers=None,
21
+ ddp_bucket_cap_mb=None,
22
+ ddp_find_unused_parameters=None,
23
+ ddp_timeout=1800,
24
+ debug=[],
25
+ deepspeed=None,
26
+ disable_tqdm=False,
27
+ dispatch_batches=None,
28
+ do_eval=True,
29
+ do_predict=False,
30
+ do_train=True,
31
+ eval_accumulation_steps=None,
32
+ eval_delay=0,
33
+ eval_do_concat_batches=True,
34
+ eval_on_start=False,
35
+ eval_steps=None,
36
+ eval_strategy=IntervalStrategy.EPOCH,
37
+ eval_use_gather_object=False,
38
+ evaluation_strategy=epoch,
39
+ fp16=False,
40
+ fp16_backend=auto,
41
+ fp16_full_eval=False,
42
+ fp16_opt_level=O1,
43
+ fsdp=[],
44
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
45
+ fsdp_min_num_params=0,
46
+ fsdp_transformer_layer_cls_to_wrap=None,
47
+ full_determinism=False,
48
+ gradient_accumulation_steps=1,
49
+ gradient_checkpointing=False,
50
+ gradient_checkpointing_kwargs=None,
51
+ greater_is_better=False,
52
+ group_by_length=False,
53
+ half_precision_backend=auto,
54
+ hub_always_push=False,
55
+ hub_model_id=None,
56
+ hub_private_repo=False,
57
+ hub_strategy=HubStrategy.EVERY_SAVE,
58
+ hub_token=<HUB_TOKEN>,
59
+ ignore_data_skip=False,
60
+ include_inputs_for_metrics=False,
61
+ include_num_input_tokens_seen=False,
62
+ include_tokens_per_second=False,
63
+ jit_mode_eval=False,
64
+ label_names=None,
65
+ label_smoothing_factor=0.0,
66
+ learning_rate=5e-05,
67
+ length_column_name=length,
68
+ load_best_model_at_end=True,
69
+ local_rank=0,
70
+ log_level=passive,
71
+ log_level_replica=warning,
72
+ log_on_each_node=True,
73
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_09-54-55_lmgpu-node-07,
74
+ logging_first_step=False,
75
+ logging_nan_inf_filter=True,
76
+ logging_steps=500,
77
+ logging_strategy=IntervalStrategy.EPOCH,
78
+ lr_scheduler_kwargs={},
79
+ lr_scheduler_type=SchedulerType.LINEAR,
80
+ max_grad_norm=1.0,
81
+ max_steps=-1,
82
+ metric_for_best_model=loss,
83
+ mp_parameters=,
84
+ neftune_noise_alpha=None,
85
+ no_cuda=False,
86
+ num_train_epochs=20.0,
87
+ optim=OptimizerNames.ADAMW_TORCH,
88
+ optim_args=None,
89
+ optim_target_modules=None,
90
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
91
+ overwrite_output_dir=False,
92
+ past_index=-1,
93
+ per_device_eval_batch_size=8,
94
+ per_device_train_batch_size=8,
95
+ prediction_loss_only=False,
96
+ push_to_hub=True,
97
+ push_to_hub_model_id=None,
98
+ push_to_hub_organization=None,
99
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
100
+ ray_scope=last,
101
+ remove_unused_columns=True,
102
+ report_to=[],
103
+ restore_callback_states_from_checkpoint=False,
104
+ resume_from_checkpoint=None,
105
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
106
+ save_on_each_node=False,
107
+ save_only_model=False,
108
+ save_safetensors=True,
109
+ save_steps=500,
110
+ save_strategy=IntervalStrategy.EPOCH,
111
+ save_total_limit=None,
112
+ seed=42,
113
+ skip_memory_metrics=True,
114
+ split_batches=None,
115
+ tf32=None,
116
+ torch_compile=False,
117
+ torch_compile_backend=None,
118
+ torch_compile_mode=None,
119
+ torch_empty_cache_steps=None,
120
+ torchdynamo=None,
121
+ tpu_metrics_debug=False,
122
+ tpu_num_cores=None,
123
+ use_cpu=False,
124
+ use_ipex=False,
125
+ use_legacy_prediction_loop=False,
126
+ use_mps_device=False,
127
+ warmup_ratio=0.0,
128
+ warmup_steps=500,
129
+ weight_decay=0.0,
130
+ )
131
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
132
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
133
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
134
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
135
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
136
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
137
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
138
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
139
+ _n_gpu=1,
140
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
141
+ adafactor=False,
142
+ adam_beta1=0.9,
143
+ adam_beta2=0.999,
144
+ adam_epsilon=1e-08,
145
+ auto_find_batch_size=False,
146
+ batch_eval_metrics=False,
147
+ bf16=False,
148
+ bf16_full_eval=False,
149
+ data_seed=None,
150
+ dataloader_drop_last=False,
151
+ dataloader_num_workers=0,
152
+ dataloader_persistent_workers=False,
153
+ dataloader_pin_memory=True,
154
+ dataloader_prefetch_factor=None,
155
+ ddp_backend=None,
156
+ ddp_broadcast_buffers=None,
157
+ ddp_bucket_cap_mb=None,
158
+ ddp_find_unused_parameters=None,
159
+ ddp_timeout=1800,
160
+ debug=[],
161
+ deepspeed=None,
162
+ disable_tqdm=False,
163
+ dispatch_batches=None,
164
+ do_eval=True,
165
+ do_predict=False,
166
+ do_train=True,
167
+ eval_accumulation_steps=None,
168
+ eval_delay=0,
169
+ eval_do_concat_batches=True,
170
+ eval_on_start=False,
171
+ eval_steps=None,
172
+ eval_strategy=IntervalStrategy.EPOCH,
173
+ eval_use_gather_object=False,
174
+ evaluation_strategy=epoch,
175
+ fp16=False,
176
+ fp16_backend=auto,
177
+ fp16_full_eval=False,
178
+ fp16_opt_level=O1,
179
+ fsdp=[],
180
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
181
+ fsdp_min_num_params=0,
182
+ fsdp_transformer_layer_cls_to_wrap=None,
183
+ full_determinism=False,
184
+ gradient_accumulation_steps=1,
185
+ gradient_checkpointing=False,
186
+ gradient_checkpointing_kwargs=None,
187
+ greater_is_better=False,
188
+ group_by_length=False,
189
+ half_precision_backend=auto,
190
+ hub_always_push=False,
191
+ hub_model_id=None,
192
+ hub_private_repo=False,
193
+ hub_strategy=HubStrategy.EVERY_SAVE,
194
+ hub_token=<HUB_TOKEN>,
195
+ ignore_data_skip=False,
196
+ include_inputs_for_metrics=False,
197
+ include_num_input_tokens_seen=False,
198
+ include_tokens_per_second=False,
199
+ jit_mode_eval=False,
200
+ label_names=None,
201
+ label_smoothing_factor=0.0,
202
+ learning_rate=5e-05,
203
+ length_column_name=length,
204
+ load_best_model_at_end=True,
205
+ local_rank=0,
206
+ log_level=passive,
207
+ log_level_replica=warning,
208
+ log_on_each_node=True,
209
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_09-59-25_lmgpu-node-07,
210
+ logging_first_step=False,
211
+ logging_nan_inf_filter=True,
212
+ logging_steps=500,
213
+ logging_strategy=IntervalStrategy.EPOCH,
214
+ lr_scheduler_kwargs={},
215
+ lr_scheduler_type=SchedulerType.LINEAR,
216
+ max_grad_norm=1.0,
217
+ max_steps=-1,
218
+ metric_for_best_model=loss,
219
+ mp_parameters=,
220
+ neftune_noise_alpha=None,
221
+ no_cuda=False,
222
+ num_train_epochs=20.0,
223
+ optim=OptimizerNames.ADAMW_TORCH,
224
+ optim_args=None,
225
+ optim_target_modules=None,
226
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
227
+ overwrite_output_dir=False,
228
+ past_index=-1,
229
+ per_device_eval_batch_size=8,
230
+ per_device_train_batch_size=8,
231
+ prediction_loss_only=False,
232
+ push_to_hub=True,
233
+ push_to_hub_model_id=None,
234
+ push_to_hub_organization=None,
235
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
236
+ ray_scope=last,
237
+ remove_unused_columns=True,
238
+ report_to=[],
239
+ restore_callback_states_from_checkpoint=False,
240
+ resume_from_checkpoint=None,
241
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
242
+ save_on_each_node=False,
243
+ save_only_model=False,
244
+ save_safetensors=True,
245
+ save_steps=500,
246
+ save_strategy=IntervalStrategy.EPOCH,
247
+ save_total_limit=None,
248
+ seed=42,
249
+ skip_memory_metrics=True,
250
+ split_batches=None,
251
+ tf32=None,
252
+ torch_compile=False,
253
+ torch_compile_backend=None,
254
+ torch_compile_mode=None,
255
+ torch_empty_cache_steps=None,
256
+ torchdynamo=None,
257
+ tpu_metrics_debug=False,
258
+ tpu_num_cores=None,
259
+ use_cpu=False,
260
+ use_ipex=False,
261
+ use_legacy_prediction_loop=False,
262
+ use_mps_device=False,
263
+ warmup_ratio=0.0,
264
+ warmup_steps=500,
265
+ weight_decay=0.0,
266
+ )
267
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
268
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
269
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
270
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
271
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
272
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
273
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
274
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
275
+ _n_gpu=1,
276
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
277
+ adafactor=False,
278
+ adam_beta1=0.9,
279
+ adam_beta2=0.999,
280
+ adam_epsilon=1e-08,
281
+ auto_find_batch_size=False,
282
+ batch_eval_metrics=False,
283
+ bf16=False,
284
+ bf16_full_eval=False,
285
+ data_seed=None,
286
+ dataloader_drop_last=False,
287
+ dataloader_num_workers=0,
288
+ dataloader_persistent_workers=False,
289
+ dataloader_pin_memory=True,
290
+ dataloader_prefetch_factor=None,
291
+ ddp_backend=None,
292
+ ddp_broadcast_buffers=None,
293
+ ddp_bucket_cap_mb=None,
294
+ ddp_find_unused_parameters=None,
295
+ ddp_timeout=1800,
296
+ debug=[],
297
+ deepspeed=None,
298
+ disable_tqdm=False,
299
+ dispatch_batches=None,
300
+ do_eval=True,
301
+ do_predict=False,
302
+ do_train=True,
303
+ eval_accumulation_steps=None,
304
+ eval_delay=0,
305
+ eval_do_concat_batches=True,
306
+ eval_on_start=False,
307
+ eval_steps=None,
308
+ eval_strategy=IntervalStrategy.EPOCH,
309
+ eval_use_gather_object=False,
310
+ evaluation_strategy=epoch,
311
+ fp16=False,
312
+ fp16_backend=auto,
313
+ fp16_full_eval=False,
314
+ fp16_opt_level=O1,
315
+ fsdp=[],
316
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
317
+ fsdp_min_num_params=0,
318
+ fsdp_transformer_layer_cls_to_wrap=None,
319
+ full_determinism=False,
320
+ gradient_accumulation_steps=1,
321
+ gradient_checkpointing=False,
322
+ gradient_checkpointing_kwargs=None,
323
+ greater_is_better=False,
324
+ group_by_length=False,
325
+ half_precision_backend=auto,
326
+ hub_always_push=False,
327
+ hub_model_id=None,
328
+ hub_private_repo=False,
329
+ hub_strategy=HubStrategy.EVERY_SAVE,
330
+ hub_token=<HUB_TOKEN>,
331
+ ignore_data_skip=False,
332
+ include_inputs_for_metrics=False,
333
+ include_num_input_tokens_seen=False,
334
+ include_tokens_per_second=False,
335
+ jit_mode_eval=False,
336
+ label_names=None,
337
+ label_smoothing_factor=0.0,
338
+ learning_rate=5e-05,
339
+ length_column_name=length,
340
+ load_best_model_at_end=True,
341
+ local_rank=0,
342
+ log_level=passive,
343
+ log_level_replica=warning,
344
+ log_on_each_node=True,
345
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_10-40-29_lmgpu-node-07,
346
+ logging_first_step=False,
347
+ logging_nan_inf_filter=True,
348
+ logging_steps=500,
349
+ logging_strategy=IntervalStrategy.EPOCH,
350
+ lr_scheduler_kwargs={},
351
+ lr_scheduler_type=SchedulerType.LINEAR,
352
+ max_grad_norm=1.0,
353
+ max_steps=-1,
354
+ metric_for_best_model=loss,
355
+ mp_parameters=,
356
+ neftune_noise_alpha=None,
357
+ no_cuda=False,
358
+ num_train_epochs=20.0,
359
+ optim=OptimizerNames.ADAMW_TORCH,
360
+ optim_args=None,
361
+ optim_target_modules=None,
362
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
363
+ overwrite_output_dir=False,
364
+ past_index=-1,
365
+ per_device_eval_batch_size=8,
366
+ per_device_train_batch_size=8,
367
+ prediction_loss_only=False,
368
+ push_to_hub=True,
369
+ push_to_hub_model_id=None,
370
+ push_to_hub_organization=None,
371
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
372
+ ray_scope=last,
373
+ remove_unused_columns=True,
374
+ report_to=[],
375
+ restore_callback_states_from_checkpoint=False,
376
+ resume_from_checkpoint=None,
377
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
378
+ save_on_each_node=False,
379
+ save_only_model=False,
380
+ save_safetensors=True,
381
+ save_steps=500,
382
+ save_strategy=IntervalStrategy.EPOCH,
383
+ save_total_limit=None,
384
+ seed=42,
385
+ skip_memory_metrics=True,
386
+ split_batches=None,
387
+ tf32=None,
388
+ torch_compile=False,
389
+ torch_compile_backend=None,
390
+ torch_compile_mode=None,
391
+ torch_empty_cache_steps=None,
392
+ torchdynamo=None,
393
+ tpu_metrics_debug=False,
394
+ tpu_num_cores=None,
395
+ use_cpu=False,
396
+ use_ipex=False,
397
+ use_legacy_prediction_loop=False,
398
+ use_mps_device=False,
399
+ warmup_ratio=0.0,
400
+ warmup_steps=500,
401
+ weight_decay=0.0,
402
+ )
403
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
404
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
405
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
406
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
407
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
408
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
409
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
410
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
411
+ _n_gpu=1,
412
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
413
+ adafactor=False,
414
+ adam_beta1=0.9,
415
+ adam_beta2=0.999,
416
+ adam_epsilon=1e-08,
417
+ auto_find_batch_size=False,
418
+ batch_eval_metrics=False,
419
+ bf16=False,
420
+ bf16_full_eval=False,
421
+ data_seed=None,
422
+ dataloader_drop_last=False,
423
+ dataloader_num_workers=0,
424
+ dataloader_persistent_workers=False,
425
+ dataloader_pin_memory=True,
426
+ dataloader_prefetch_factor=None,
427
+ ddp_backend=None,
428
+ ddp_broadcast_buffers=None,
429
+ ddp_bucket_cap_mb=None,
430
+ ddp_find_unused_parameters=None,
431
+ ddp_timeout=1800,
432
+ debug=[],
433
+ deepspeed=None,
434
+ disable_tqdm=False,
435
+ dispatch_batches=None,
436
+ do_eval=True,
437
+ do_predict=False,
438
+ do_train=True,
439
+ eval_accumulation_steps=None,
440
+ eval_delay=0,
441
+ eval_do_concat_batches=True,
442
+ eval_on_start=False,
443
+ eval_steps=None,
444
+ eval_strategy=IntervalStrategy.EPOCH,
445
+ eval_use_gather_object=False,
446
+ evaluation_strategy=epoch,
447
+ fp16=False,
448
+ fp16_backend=auto,
449
+ fp16_full_eval=False,
450
+ fp16_opt_level=O1,
451
+ fsdp=[],
452
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
453
+ fsdp_min_num_params=0,
454
+ fsdp_transformer_layer_cls_to_wrap=None,
455
+ full_determinism=False,
456
+ gradient_accumulation_steps=1,
457
+ gradient_checkpointing=False,
458
+ gradient_checkpointing_kwargs=None,
459
+ greater_is_better=False,
460
+ group_by_length=False,
461
+ half_precision_backend=auto,
462
+ hub_always_push=False,
463
+ hub_model_id=None,
464
+ hub_private_repo=False,
465
+ hub_strategy=HubStrategy.EVERY_SAVE,
466
+ hub_token=<HUB_TOKEN>,
467
+ ignore_data_skip=False,
468
+ include_inputs_for_metrics=False,
469
+ include_num_input_tokens_seen=False,
470
+ include_tokens_per_second=False,
471
+ jit_mode_eval=False,
472
+ label_names=None,
473
+ label_smoothing_factor=0.0,
474
+ learning_rate=5e-05,
475
+ length_column_name=length,
476
+ load_best_model_at_end=True,
477
+ local_rank=0,
478
+ log_level=passive,
479
+ log_level_replica=warning,
480
+ log_on_each_node=True,
481
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_11-01-55_lmgpu-node-07,
482
+ logging_first_step=False,
483
+ logging_nan_inf_filter=True,
484
+ logging_steps=500,
485
+ logging_strategy=IntervalStrategy.EPOCH,
486
+ lr_scheduler_kwargs={},
487
+ lr_scheduler_type=SchedulerType.LINEAR,
488
+ max_grad_norm=1.0,
489
+ max_steps=-1,
490
+ metric_for_best_model=loss,
491
+ mp_parameters=,
492
+ neftune_noise_alpha=None,
493
+ no_cuda=False,
494
+ num_train_epochs=20.0,
495
+ optim=OptimizerNames.ADAMW_TORCH,
496
+ optim_args=None,
497
+ optim_target_modules=None,
498
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
499
+ overwrite_output_dir=False,
500
+ past_index=-1,
501
+ per_device_eval_batch_size=8,
502
+ per_device_train_batch_size=8,
503
+ prediction_loss_only=False,
504
+ push_to_hub=True,
505
+ push_to_hub_model_id=None,
506
+ push_to_hub_organization=None,
507
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
508
+ ray_scope=last,
509
+ remove_unused_columns=True,
510
+ report_to=[],
511
+ restore_callback_states_from_checkpoint=False,
512
+ resume_from_checkpoint=None,
513
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
514
+ save_on_each_node=False,
515
+ save_only_model=False,
516
+ save_safetensors=True,
517
+ save_steps=500,
518
+ save_strategy=IntervalStrategy.EPOCH,
519
+ save_total_limit=None,
520
+ seed=42,
521
+ skip_memory_metrics=True,
522
+ split_batches=None,
523
+ tf32=None,
524
+ torch_compile=False,
525
+ torch_compile_backend=None,
526
+ torch_compile_mode=None,
527
+ torch_empty_cache_steps=None,
528
+ torchdynamo=None,
529
+ tpu_metrics_debug=False,
530
+ tpu_num_cores=None,
531
+ use_cpu=False,
532
+ use_ipex=False,
533
+ use_legacy_prediction_loop=False,
534
+ use_mps_device=False,
535
+ warmup_ratio=0.0,
536
+ warmup_steps=500,
537
+ weight_decay=0.0,
538
+ )
539
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
540
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
541
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
542
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
543
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
544
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
545
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
546
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
547
+ _n_gpu=1,
548
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
549
+ adafactor=False,
550
+ adam_beta1=0.9,
551
+ adam_beta2=0.999,
552
+ adam_epsilon=1e-08,
553
+ auto_find_batch_size=False,
554
+ batch_eval_metrics=False,
555
+ bf16=False,
556
+ bf16_full_eval=False,
557
+ data_seed=None,
558
+ dataloader_drop_last=False,
559
+ dataloader_num_workers=0,
560
+ dataloader_persistent_workers=False,
561
+ dataloader_pin_memory=True,
562
+ dataloader_prefetch_factor=None,
563
+ ddp_backend=None,
564
+ ddp_broadcast_buffers=None,
565
+ ddp_bucket_cap_mb=None,
566
+ ddp_find_unused_parameters=None,
567
+ ddp_timeout=1800,
568
+ debug=[],
569
+ deepspeed=None,
570
+ disable_tqdm=False,
571
+ dispatch_batches=None,
572
+ do_eval=True,
573
+ do_predict=False,
574
+ do_train=True,
575
+ eval_accumulation_steps=None,
576
+ eval_delay=0,
577
+ eval_do_concat_batches=True,
578
+ eval_on_start=False,
579
+ eval_steps=None,
580
+ eval_strategy=IntervalStrategy.EPOCH,
581
+ eval_use_gather_object=False,
582
+ evaluation_strategy=epoch,
583
+ fp16=False,
584
+ fp16_backend=auto,
585
+ fp16_full_eval=False,
586
+ fp16_opt_level=O1,
587
+ fsdp=[],
588
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
589
+ fsdp_min_num_params=0,
590
+ fsdp_transformer_layer_cls_to_wrap=None,
591
+ full_determinism=False,
592
+ gradient_accumulation_steps=1,
593
+ gradient_checkpointing=False,
594
+ gradient_checkpointing_kwargs=None,
595
+ greater_is_better=False,
596
+ group_by_length=False,
597
+ half_precision_backend=auto,
598
+ hub_always_push=False,
599
+ hub_model_id=None,
600
+ hub_private_repo=False,
601
+ hub_strategy=HubStrategy.EVERY_SAVE,
602
+ hub_token=<HUB_TOKEN>,
603
+ ignore_data_skip=False,
604
+ include_inputs_for_metrics=False,
605
+ include_num_input_tokens_seen=False,
606
+ include_tokens_per_second=False,
607
+ jit_mode_eval=False,
608
+ label_names=None,
609
+ label_smoothing_factor=0.0,
610
+ learning_rate=5e-05,
611
+ length_column_name=length,
612
+ load_best_model_at_end=True,
613
+ local_rank=0,
614
+ log_level=passive,
615
+ log_level_replica=warning,
616
+ log_on_each_node=True,
617
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_11-04-55_lmgpu-node-07,
618
+ logging_first_step=False,
619
+ logging_nan_inf_filter=True,
620
+ logging_steps=500,
621
+ logging_strategy=IntervalStrategy.EPOCH,
622
+ lr_scheduler_kwargs={},
623
+ lr_scheduler_type=SchedulerType.LINEAR,
624
+ max_grad_norm=1.0,
625
+ max_steps=-1,
626
+ metric_for_best_model=loss,
627
+ mp_parameters=,
628
+ neftune_noise_alpha=None,
629
+ no_cuda=False,
630
+ num_train_epochs=20.0,
631
+ optim=OptimizerNames.ADAMW_TORCH,
632
+ optim_args=None,
633
+ optim_target_modules=None,
634
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
635
+ overwrite_output_dir=False,
636
+ past_index=-1,
637
+ per_device_eval_batch_size=8,
638
+ per_device_train_batch_size=8,
639
+ prediction_loss_only=False,
640
+ push_to_hub=True,
641
+ push_to_hub_model_id=None,
642
+ push_to_hub_organization=None,
643
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
644
+ ray_scope=last,
645
+ remove_unused_columns=True,
646
+ report_to=[],
647
+ restore_callback_states_from_checkpoint=False,
648
+ resume_from_checkpoint=None,
649
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
650
+ save_on_each_node=False,
651
+ save_only_model=False,
652
+ save_safetensors=True,
653
+ save_steps=500,
654
+ save_strategy=IntervalStrategy.EPOCH,
655
+ save_total_limit=None,
656
+ seed=42,
657
+ skip_memory_metrics=True,
658
+ split_batches=None,
659
+ tf32=None,
660
+ torch_compile=False,
661
+ torch_compile_backend=None,
662
+ torch_compile_mode=None,
663
+ torch_empty_cache_steps=None,
664
+ torchdynamo=None,
665
+ tpu_metrics_debug=False,
666
+ tpu_num_cores=None,
667
+ use_cpu=False,
668
+ use_ipex=False,
669
+ use_legacy_prediction_loop=False,
670
+ use_mps_device=False,
671
+ warmup_ratio=0.0,
672
+ warmup_steps=500,
673
+ weight_decay=0.0,
674
+ )
675
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
676
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
677
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
678
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
679
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
680
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
681
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
682
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
683
+ _n_gpu=1,
684
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
685
+ adafactor=False,
686
+ adam_beta1=0.9,
687
+ adam_beta2=0.999,
688
+ adam_epsilon=1e-08,
689
+ auto_find_batch_size=False,
690
+ batch_eval_metrics=False,
691
+ bf16=False,
692
+ bf16_full_eval=False,
693
+ data_seed=None,
694
+ dataloader_drop_last=False,
695
+ dataloader_num_workers=0,
696
+ dataloader_persistent_workers=False,
697
+ dataloader_pin_memory=True,
698
+ dataloader_prefetch_factor=None,
699
+ ddp_backend=None,
700
+ ddp_broadcast_buffers=None,
701
+ ddp_bucket_cap_mb=None,
702
+ ddp_find_unused_parameters=None,
703
+ ddp_timeout=1800,
704
+ debug=[],
705
+ deepspeed=None,
706
+ disable_tqdm=False,
707
+ dispatch_batches=None,
708
+ do_eval=True,
709
+ do_predict=False,
710
+ do_train=True,
711
+ eval_accumulation_steps=None,
712
+ eval_delay=0,
713
+ eval_do_concat_batches=True,
714
+ eval_on_start=False,
715
+ eval_steps=None,
716
+ eval_strategy=IntervalStrategy.EPOCH,
717
+ eval_use_gather_object=False,
718
+ evaluation_strategy=epoch,
719
+ fp16=False,
720
+ fp16_backend=auto,
721
+ fp16_full_eval=False,
722
+ fp16_opt_level=O1,
723
+ fsdp=[],
724
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
725
+ fsdp_min_num_params=0,
726
+ fsdp_transformer_layer_cls_to_wrap=None,
727
+ full_determinism=False,
728
+ gradient_accumulation_steps=1,
729
+ gradient_checkpointing=False,
730
+ gradient_checkpointing_kwargs=None,
731
+ greater_is_better=False,
732
+ group_by_length=False,
733
+ half_precision_backend=auto,
734
+ hub_always_push=False,
735
+ hub_model_id=None,
736
+ hub_private_repo=False,
737
+ hub_strategy=HubStrategy.EVERY_SAVE,
738
+ hub_token=<HUB_TOKEN>,
739
+ ignore_data_skip=False,
740
+ include_inputs_for_metrics=False,
741
+ include_num_input_tokens_seen=False,
742
+ include_tokens_per_second=False,
743
+ jit_mode_eval=False,
744
+ label_names=None,
745
+ label_smoothing_factor=0.0,
746
+ learning_rate=5e-05,
747
+ length_column_name=length,
748
+ load_best_model_at_end=True,
749
+ local_rank=0,
750
+ log_level=passive,
751
+ log_level_replica=warning,
752
+ log_on_each_node=True,
753
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_11-40-25_lmgpu-node-07,
754
+ logging_first_step=False,
755
+ logging_nan_inf_filter=True,
756
+ logging_steps=500,
757
+ logging_strategy=IntervalStrategy.EPOCH,
758
+ lr_scheduler_kwargs={},
759
+ lr_scheduler_type=SchedulerType.LINEAR,
760
+ max_grad_norm=1.0,
761
+ max_steps=-1,
762
+ metric_for_best_model=loss,
763
+ mp_parameters=,
764
+ neftune_noise_alpha=None,
765
+ no_cuda=False,
766
+ num_train_epochs=20.0,
767
+ optim=OptimizerNames.ADAMW_TORCH,
768
+ optim_args=None,
769
+ optim_target_modules=None,
770
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
771
+ overwrite_output_dir=False,
772
+ past_index=-1,
773
+ per_device_eval_batch_size=8,
774
+ per_device_train_batch_size=8,
775
+ prediction_loss_only=False,
776
+ push_to_hub=True,
777
+ push_to_hub_model_id=None,
778
+ push_to_hub_organization=None,
779
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
780
+ ray_scope=last,
781
+ remove_unused_columns=True,
782
+ report_to=[],
783
+ restore_callback_states_from_checkpoint=False,
784
+ resume_from_checkpoint=None,
785
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
786
+ save_on_each_node=False,
787
+ save_only_model=False,
788
+ save_safetensors=True,
789
+ save_steps=500,
790
+ save_strategy=IntervalStrategy.EPOCH,
791
+ save_total_limit=None,
792
+ seed=42,
793
+ skip_memory_metrics=True,
794
+ split_batches=None,
795
+ tf32=None,
796
+ torch_compile=False,
797
+ torch_compile_backend=None,
798
+ torch_compile_mode=None,
799
+ torch_empty_cache_steps=None,
800
+ torchdynamo=None,
801
+ tpu_metrics_debug=False,
802
+ tpu_num_cores=None,
803
+ use_cpu=False,
804
+ use_ipex=False,
805
+ use_legacy_prediction_loop=False,
806
+ use_mps_device=False,
807
+ warmup_ratio=0.0,
808
+ warmup_steps=500,
809
+ weight_decay=0.0,
810
+ )
811
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
812
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
813
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
814
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
815
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
816
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
817
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
818
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
819
+ _n_gpu=1,
820
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
821
+ adafactor=False,
822
+ adam_beta1=0.9,
823
+ adam_beta2=0.999,
824
+ adam_epsilon=1e-08,
825
+ auto_find_batch_size=False,
826
+ batch_eval_metrics=False,
827
+ bf16=False,
828
+ bf16_full_eval=False,
829
+ data_seed=None,
830
+ dataloader_drop_last=False,
831
+ dataloader_num_workers=0,
832
+ dataloader_persistent_workers=False,
833
+ dataloader_pin_memory=True,
834
+ dataloader_prefetch_factor=None,
835
+ ddp_backend=None,
836
+ ddp_broadcast_buffers=None,
837
+ ddp_bucket_cap_mb=None,
838
+ ddp_find_unused_parameters=None,
839
+ ddp_timeout=1800,
840
+ debug=[],
841
+ deepspeed=None,
842
+ disable_tqdm=False,
843
+ dispatch_batches=None,
844
+ do_eval=True,
845
+ do_predict=False,
846
+ do_train=True,
847
+ eval_accumulation_steps=None,
848
+ eval_delay=0,
849
+ eval_do_concat_batches=True,
850
+ eval_on_start=False,
851
+ eval_steps=None,
852
+ eval_strategy=IntervalStrategy.EPOCH,
853
+ eval_use_gather_object=False,
854
+ evaluation_strategy=epoch,
855
+ fp16=False,
856
+ fp16_backend=auto,
857
+ fp16_full_eval=False,
858
+ fp16_opt_level=O1,
859
+ fsdp=[],
860
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
861
+ fsdp_min_num_params=0,
862
+ fsdp_transformer_layer_cls_to_wrap=None,
863
+ full_determinism=False,
864
+ gradient_accumulation_steps=1,
865
+ gradient_checkpointing=False,
866
+ gradient_checkpointing_kwargs=None,
867
+ greater_is_better=False,
868
+ group_by_length=False,
869
+ half_precision_backend=auto,
870
+ hub_always_push=False,
871
+ hub_model_id=None,
872
+ hub_private_repo=False,
873
+ hub_strategy=HubStrategy.EVERY_SAVE,
874
+ hub_token=<HUB_TOKEN>,
875
+ ignore_data_skip=False,
876
+ include_inputs_for_metrics=False,
877
+ include_num_input_tokens_seen=False,
878
+ include_tokens_per_second=False,
879
+ jit_mode_eval=False,
880
+ label_names=None,
881
+ label_smoothing_factor=0.0,
882
+ learning_rate=5e-05,
883
+ length_column_name=length,
884
+ load_best_model_at_end=True,
885
+ local_rank=0,
886
+ log_level=passive,
887
+ log_level_replica=warning,
888
+ log_on_each_node=True,
889
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-01-55_lmgpu-node-07,
890
+ logging_first_step=False,
891
+ logging_nan_inf_filter=True,
892
+ logging_steps=500,
893
+ logging_strategy=IntervalStrategy.EPOCH,
894
+ lr_scheduler_kwargs={},
895
+ lr_scheduler_type=SchedulerType.LINEAR,
896
+ max_grad_norm=1.0,
897
+ max_steps=-1,
898
+ metric_for_best_model=loss,
899
+ mp_parameters=,
900
+ neftune_noise_alpha=None,
901
+ no_cuda=False,
902
+ num_train_epochs=20.0,
903
+ optim=OptimizerNames.ADAMW_TORCH,
904
+ optim_args=None,
905
+ optim_target_modules=None,
906
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
907
+ overwrite_output_dir=False,
908
+ past_index=-1,
909
+ per_device_eval_batch_size=8,
910
+ per_device_train_batch_size=8,
911
+ prediction_loss_only=False,
912
+ push_to_hub=True,
913
+ push_to_hub_model_id=None,
914
+ push_to_hub_organization=None,
915
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
916
+ ray_scope=last,
917
+ remove_unused_columns=True,
918
+ report_to=[],
919
+ restore_callback_states_from_checkpoint=False,
920
+ resume_from_checkpoint=None,
921
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
922
+ save_on_each_node=False,
923
+ save_only_model=False,
924
+ save_safetensors=True,
925
+ save_steps=500,
926
+ save_strategy=IntervalStrategy.EPOCH,
927
+ save_total_limit=None,
928
+ seed=42,
929
+ skip_memory_metrics=True,
930
+ split_batches=None,
931
+ tf32=None,
932
+ torch_compile=False,
933
+ torch_compile_backend=None,
934
+ torch_compile_mode=None,
935
+ torch_empty_cache_steps=None,
936
+ torchdynamo=None,
937
+ tpu_metrics_debug=False,
938
+ tpu_num_cores=None,
939
+ use_cpu=False,
940
+ use_ipex=False,
941
+ use_legacy_prediction_loop=False,
942
+ use_mps_device=False,
943
+ warmup_ratio=0.0,
944
+ warmup_steps=500,
945
+ weight_decay=0.0,
946
+ )
947
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
948
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
949
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
950
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
951
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
952
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
953
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
954
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
955
+ _n_gpu=1,
956
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
957
+ adafactor=False,
958
+ adam_beta1=0.9,
959
+ adam_beta2=0.999,
960
+ adam_epsilon=1e-08,
961
+ auto_find_batch_size=False,
962
+ batch_eval_metrics=False,
963
+ bf16=False,
964
+ bf16_full_eval=False,
965
+ data_seed=None,
966
+ dataloader_drop_last=False,
967
+ dataloader_num_workers=0,
968
+ dataloader_persistent_workers=False,
969
+ dataloader_pin_memory=True,
970
+ dataloader_prefetch_factor=None,
971
+ ddp_backend=None,
972
+ ddp_broadcast_buffers=None,
973
+ ddp_bucket_cap_mb=None,
974
+ ddp_find_unused_parameters=None,
975
+ ddp_timeout=1800,
976
+ debug=[],
977
+ deepspeed=None,
978
+ disable_tqdm=False,
979
+ dispatch_batches=None,
980
+ do_eval=True,
981
+ do_predict=False,
982
+ do_train=True,
983
+ eval_accumulation_steps=None,
984
+ eval_delay=0,
985
+ eval_do_concat_batches=True,
986
+ eval_on_start=False,
987
+ eval_steps=None,
988
+ eval_strategy=IntervalStrategy.EPOCH,
989
+ eval_use_gather_object=False,
990
+ evaluation_strategy=epoch,
991
+ fp16=False,
992
+ fp16_backend=auto,
993
+ fp16_full_eval=False,
994
+ fp16_opt_level=O1,
995
+ fsdp=[],
996
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
997
+ fsdp_min_num_params=0,
998
+ fsdp_transformer_layer_cls_to_wrap=None,
999
+ full_determinism=False,
1000
+ gradient_accumulation_steps=1,
1001
+ gradient_checkpointing=False,
1002
+ gradient_checkpointing_kwargs=None,
1003
+ greater_is_better=False,
1004
+ group_by_length=False,
1005
+ half_precision_backend=auto,
1006
+ hub_always_push=False,
1007
+ hub_model_id=None,
1008
+ hub_private_repo=False,
1009
+ hub_strategy=HubStrategy.EVERY_SAVE,
1010
+ hub_token=<HUB_TOKEN>,
1011
+ ignore_data_skip=False,
1012
+ include_inputs_for_metrics=False,
1013
+ include_num_input_tokens_seen=False,
1014
+ include_tokens_per_second=False,
1015
+ jit_mode_eval=False,
1016
+ label_names=None,
1017
+ label_smoothing_factor=0.0,
1018
+ learning_rate=5e-05,
1019
+ length_column_name=length,
1020
+ load_best_model_at_end=True,
1021
+ local_rank=0,
1022
+ log_level=passive,
1023
+ log_level_replica=warning,
1024
+ log_on_each_node=True,
1025
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-09-55_lmgpu-node-07,
1026
+ logging_first_step=False,
1027
+ logging_nan_inf_filter=True,
1028
+ logging_steps=500,
1029
+ logging_strategy=IntervalStrategy.EPOCH,
1030
+ lr_scheduler_kwargs={},
1031
+ lr_scheduler_type=SchedulerType.LINEAR,
1032
+ max_grad_norm=1.0,
1033
+ max_steps=-1,
1034
+ metric_for_best_model=loss,
1035
+ mp_parameters=,
1036
+ neftune_noise_alpha=None,
1037
+ no_cuda=False,
1038
+ num_train_epochs=20.0,
1039
+ optim=OptimizerNames.ADAMW_TORCH,
1040
+ optim_args=None,
1041
+ optim_target_modules=None,
1042
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1043
+ overwrite_output_dir=False,
1044
+ past_index=-1,
1045
+ per_device_eval_batch_size=8,
1046
+ per_device_train_batch_size=8,
1047
+ prediction_loss_only=False,
1048
+ push_to_hub=True,
1049
+ push_to_hub_model_id=None,
1050
+ push_to_hub_organization=None,
1051
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1052
+ ray_scope=last,
1053
+ remove_unused_columns=True,
1054
+ report_to=[],
1055
+ restore_callback_states_from_checkpoint=False,
1056
+ resume_from_checkpoint=None,
1057
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1058
+ save_on_each_node=False,
1059
+ save_only_model=False,
1060
+ save_safetensors=True,
1061
+ save_steps=500,
1062
+ save_strategy=IntervalStrategy.EPOCH,
1063
+ save_total_limit=None,
1064
+ seed=42,
1065
+ skip_memory_metrics=True,
1066
+ split_batches=None,
1067
+ tf32=None,
1068
+ torch_compile=False,
1069
+ torch_compile_backend=None,
1070
+ torch_compile_mode=None,
1071
+ torch_empty_cache_steps=None,
1072
+ torchdynamo=None,
1073
+ tpu_metrics_debug=False,
1074
+ tpu_num_cores=None,
1075
+ use_cpu=False,
1076
+ use_ipex=False,
1077
+ use_legacy_prediction_loop=False,
1078
+ use_mps_device=False,
1079
+ warmup_ratio=0.0,
1080
+ warmup_steps=500,
1081
+ weight_decay=0.0,
1082
+ )
1083
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1084
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1085
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1086
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1087
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1088
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1089
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
1090
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
1091
+ _n_gpu=1,
1092
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
1093
+ adafactor=False,
1094
+ adam_beta1=0.9,
1095
+ adam_beta2=0.999,
1096
+ adam_epsilon=1e-08,
1097
+ auto_find_batch_size=False,
1098
+ batch_eval_metrics=False,
1099
+ bf16=False,
1100
+ bf16_full_eval=False,
1101
+ data_seed=None,
1102
+ dataloader_drop_last=False,
1103
+ dataloader_num_workers=0,
1104
+ dataloader_persistent_workers=False,
1105
+ dataloader_pin_memory=True,
1106
+ dataloader_prefetch_factor=None,
1107
+ ddp_backend=None,
1108
+ ddp_broadcast_buffers=None,
1109
+ ddp_bucket_cap_mb=None,
1110
+ ddp_find_unused_parameters=None,
1111
+ ddp_timeout=1800,
1112
+ debug=[],
1113
+ deepspeed=None,
1114
+ disable_tqdm=False,
1115
+ dispatch_batches=None,
1116
+ do_eval=True,
1117
+ do_predict=False,
1118
+ do_train=True,
1119
+ eval_accumulation_steps=None,
1120
+ eval_delay=0,
1121
+ eval_do_concat_batches=True,
1122
+ eval_on_start=False,
1123
+ eval_steps=None,
1124
+ eval_strategy=IntervalStrategy.EPOCH,
1125
+ eval_use_gather_object=False,
1126
+ evaluation_strategy=epoch,
1127
+ fp16=False,
1128
+ fp16_backend=auto,
1129
+ fp16_full_eval=False,
1130
+ fp16_opt_level=O1,
1131
+ fsdp=[],
1132
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
1133
+ fsdp_min_num_params=0,
1134
+ fsdp_transformer_layer_cls_to_wrap=None,
1135
+ full_determinism=False,
1136
+ gradient_accumulation_steps=1,
1137
+ gradient_checkpointing=False,
1138
+ gradient_checkpointing_kwargs=None,
1139
+ greater_is_better=False,
1140
+ group_by_length=False,
1141
+ half_precision_backend=auto,
1142
+ hub_always_push=False,
1143
+ hub_model_id=None,
1144
+ hub_private_repo=False,
1145
+ hub_strategy=HubStrategy.EVERY_SAVE,
1146
+ hub_token=<HUB_TOKEN>,
1147
+ ignore_data_skip=False,
1148
+ include_inputs_for_metrics=False,
1149
+ include_num_input_tokens_seen=False,
1150
+ include_tokens_per_second=False,
1151
+ jit_mode_eval=False,
1152
+ label_names=None,
1153
+ label_smoothing_factor=0.0,
1154
+ learning_rate=5e-05,
1155
+ length_column_name=length,
1156
+ load_best_model_at_end=True,
1157
+ local_rank=0,
1158
+ log_level=passive,
1159
+ log_level_replica=warning,
1160
+ log_on_each_node=True,
1161
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-14-55_lmgpu-node-07,
1162
+ logging_first_step=False,
1163
+ logging_nan_inf_filter=True,
1164
+ logging_steps=500,
1165
+ logging_strategy=IntervalStrategy.EPOCH,
1166
+ lr_scheduler_kwargs={},
1167
+ lr_scheduler_type=SchedulerType.LINEAR,
1168
+ max_grad_norm=1.0,
1169
+ max_steps=-1,
1170
+ metric_for_best_model=loss,
1171
+ mp_parameters=,
1172
+ neftune_noise_alpha=None,
1173
+ no_cuda=False,
1174
+ num_train_epochs=20.0,
1175
+ optim=OptimizerNames.ADAMW_TORCH,
1176
+ optim_args=None,
1177
+ optim_target_modules=None,
1178
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1179
+ overwrite_output_dir=False,
1180
+ past_index=-1,
1181
+ per_device_eval_batch_size=8,
1182
+ per_device_train_batch_size=8,
1183
+ prediction_loss_only=False,
1184
+ push_to_hub=True,
1185
+ push_to_hub_model_id=None,
1186
+ push_to_hub_organization=None,
1187
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1188
+ ray_scope=last,
1189
+ remove_unused_columns=True,
1190
+ report_to=[],
1191
+ restore_callback_states_from_checkpoint=False,
1192
+ resume_from_checkpoint=None,
1193
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1194
+ save_on_each_node=False,
1195
+ save_only_model=False,
1196
+ save_safetensors=True,
1197
+ save_steps=500,
1198
+ save_strategy=IntervalStrategy.EPOCH,
1199
+ save_total_limit=None,
1200
+ seed=42,
1201
+ skip_memory_metrics=True,
1202
+ split_batches=None,
1203
+ tf32=None,
1204
+ torch_compile=False,
1205
+ torch_compile_backend=None,
1206
+ torch_compile_mode=None,
1207
+ torch_empty_cache_steps=None,
1208
+ torchdynamo=None,
1209
+ tpu_metrics_debug=False,
1210
+ tpu_num_cores=None,
1211
+ use_cpu=False,
1212
+ use_ipex=False,
1213
+ use_legacy_prediction_loop=False,
1214
+ use_mps_device=False,
1215
+ warmup_ratio=0.0,
1216
+ warmup_steps=500,
1217
+ weight_decay=0.0,
1218
+ )
1219
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1220
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1221
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1222
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1223
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1224
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1225
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
1226
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
1227
+ _n_gpu=1,
1228
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
1229
+ adafactor=False,
1230
+ adam_beta1=0.9,
1231
+ adam_beta2=0.999,
1232
+ adam_epsilon=1e-08,
1233
+ auto_find_batch_size=False,
1234
+ batch_eval_metrics=False,
1235
+ bf16=False,
1236
+ bf16_full_eval=False,
1237
+ data_seed=None,
1238
+ dataloader_drop_last=False,
1239
+ dataloader_num_workers=0,
1240
+ dataloader_persistent_workers=False,
1241
+ dataloader_pin_memory=True,
1242
+ dataloader_prefetch_factor=None,
1243
+ ddp_backend=None,
1244
+ ddp_broadcast_buffers=None,
1245
+ ddp_bucket_cap_mb=None,
1246
+ ddp_find_unused_parameters=None,
1247
+ ddp_timeout=1800,
1248
+ debug=[],
1249
+ deepspeed=None,
1250
+ disable_tqdm=False,
1251
+ dispatch_batches=None,
1252
+ do_eval=True,
1253
+ do_predict=False,
1254
+ do_train=True,
1255
+ eval_accumulation_steps=None,
1256
+ eval_delay=0,
1257
+ eval_do_concat_batches=True,
1258
+ eval_on_start=False,
1259
+ eval_steps=None,
1260
+ eval_strategy=IntervalStrategy.EPOCH,
1261
+ eval_use_gather_object=False,
1262
+ evaluation_strategy=epoch,
1263
+ fp16=False,
1264
+ fp16_backend=auto,
1265
+ fp16_full_eval=False,
1266
+ fp16_opt_level=O1,
1267
+ fsdp=[],
1268
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
1269
+ fsdp_min_num_params=0,
1270
+ fsdp_transformer_layer_cls_to_wrap=None,
1271
+ full_determinism=False,
1272
+ gradient_accumulation_steps=1,
1273
+ gradient_checkpointing=False,
1274
+ gradient_checkpointing_kwargs=None,
1275
+ greater_is_better=False,
1276
+ group_by_length=False,
1277
+ half_precision_backend=auto,
1278
+ hub_always_push=False,
1279
+ hub_model_id=None,
1280
+ hub_private_repo=False,
1281
+ hub_strategy=HubStrategy.EVERY_SAVE,
1282
+ hub_token=<HUB_TOKEN>,
1283
+ ignore_data_skip=False,
1284
+ include_inputs_for_metrics=False,
1285
+ include_num_input_tokens_seen=False,
1286
+ include_tokens_per_second=False,
1287
+ jit_mode_eval=False,
1288
+ label_names=None,
1289
+ label_smoothing_factor=0.0,
1290
+ learning_rate=5e-05,
1291
+ length_column_name=length,
1292
+ load_best_model_at_end=True,
1293
+ local_rank=0,
1294
+ log_level=passive,
1295
+ log_level_replica=warning,
1296
+ log_on_each_node=True,
1297
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-17-25_lmgpu-node-07,
1298
+ logging_first_step=False,
1299
+ logging_nan_inf_filter=True,
1300
+ logging_steps=500,
1301
+ logging_strategy=IntervalStrategy.EPOCH,
1302
+ lr_scheduler_kwargs={},
1303
+ lr_scheduler_type=SchedulerType.LINEAR,
1304
+ max_grad_norm=1.0,
1305
+ max_steps=-1,
1306
+ metric_for_best_model=loss,
1307
+ mp_parameters=,
1308
+ neftune_noise_alpha=None,
1309
+ no_cuda=False,
1310
+ num_train_epochs=20.0,
1311
+ optim=OptimizerNames.ADAMW_TORCH,
1312
+ optim_args=None,
1313
+ optim_target_modules=None,
1314
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1315
+ overwrite_output_dir=False,
1316
+ past_index=-1,
1317
+ per_device_eval_batch_size=8,
1318
+ per_device_train_batch_size=8,
1319
+ prediction_loss_only=False,
1320
+ push_to_hub=True,
1321
+ push_to_hub_model_id=None,
1322
+ push_to_hub_organization=None,
1323
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1324
+ ray_scope=last,
1325
+ remove_unused_columns=True,
1326
+ report_to=[],
1327
+ restore_callback_states_from_checkpoint=False,
1328
+ resume_from_checkpoint=None,
1329
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1330
+ save_on_each_node=False,
1331
+ save_only_model=False,
1332
+ save_safetensors=True,
1333
+ save_steps=500,
1334
+ save_strategy=IntervalStrategy.EPOCH,
1335
+ save_total_limit=None,
1336
+ seed=42,
1337
+ skip_memory_metrics=True,
1338
+ split_batches=None,
1339
+ tf32=None,
1340
+ torch_compile=False,
1341
+ torch_compile_backend=None,
1342
+ torch_compile_mode=None,
1343
+ torch_empty_cache_steps=None,
1344
+ torchdynamo=None,
1345
+ tpu_metrics_debug=False,
1346
+ tpu_num_cores=None,
1347
+ use_cpu=False,
1348
+ use_ipex=False,
1349
+ use_legacy_prediction_loop=False,
1350
+ use_mps_device=False,
1351
+ warmup_ratio=0.0,
1352
+ warmup_steps=500,
1353
+ weight_decay=0.0,
1354
+ )
1355
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1356
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1357
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1358
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1359
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1360
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1361
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
1362
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
1363
+ _n_gpu=1,
1364
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
1365
+ adafactor=False,
1366
+ adam_beta1=0.9,
1367
+ adam_beta2=0.999,
1368
+ adam_epsilon=1e-08,
1369
+ auto_find_batch_size=False,
1370
+ batch_eval_metrics=False,
1371
+ bf16=False,
1372
+ bf16_full_eval=False,
1373
+ data_seed=None,
1374
+ dataloader_drop_last=False,
1375
+ dataloader_num_workers=0,
1376
+ dataloader_persistent_workers=False,
1377
+ dataloader_pin_memory=True,
1378
+ dataloader_prefetch_factor=None,
1379
+ ddp_backend=None,
1380
+ ddp_broadcast_buffers=None,
1381
+ ddp_bucket_cap_mb=None,
1382
+ ddp_find_unused_parameters=None,
1383
+ ddp_timeout=1800,
1384
+ debug=[],
1385
+ deepspeed=None,
1386
+ disable_tqdm=False,
1387
+ dispatch_batches=None,
1388
+ do_eval=True,
1389
+ do_predict=False,
1390
+ do_train=True,
1391
+ eval_accumulation_steps=None,
1392
+ eval_delay=0,
1393
+ eval_do_concat_batches=True,
1394
+ eval_on_start=False,
1395
+ eval_steps=None,
1396
+ eval_strategy=IntervalStrategy.EPOCH,
1397
+ eval_use_gather_object=False,
1398
+ evaluation_strategy=epoch,
1399
+ fp16=False,
1400
+ fp16_backend=auto,
1401
+ fp16_full_eval=False,
1402
+ fp16_opt_level=O1,
1403
+ fsdp=[],
1404
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
1405
+ fsdp_min_num_params=0,
1406
+ fsdp_transformer_layer_cls_to_wrap=None,
1407
+ full_determinism=False,
1408
+ gradient_accumulation_steps=1,
1409
+ gradient_checkpointing=False,
1410
+ gradient_checkpointing_kwargs=None,
1411
+ greater_is_better=False,
1412
+ group_by_length=False,
1413
+ half_precision_backend=auto,
1414
+ hub_always_push=False,
1415
+ hub_model_id=None,
1416
+ hub_private_repo=False,
1417
+ hub_strategy=HubStrategy.EVERY_SAVE,
1418
+ hub_token=<HUB_TOKEN>,
1419
+ ignore_data_skip=False,
1420
+ include_inputs_for_metrics=False,
1421
+ include_num_input_tokens_seen=False,
1422
+ include_tokens_per_second=False,
1423
+ jit_mode_eval=False,
1424
+ label_names=None,
1425
+ label_smoothing_factor=0.0,
1426
+ learning_rate=5e-05,
1427
+ length_column_name=length,
1428
+ load_best_model_at_end=True,
1429
+ local_rank=0,
1430
+ log_level=passive,
1431
+ log_level_replica=warning,
1432
+ log_on_each_node=True,
1433
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-20-55_lmgpu-node-07,
1434
+ logging_first_step=False,
1435
+ logging_nan_inf_filter=True,
1436
+ logging_steps=500,
1437
+ logging_strategy=IntervalStrategy.EPOCH,
1438
+ lr_scheduler_kwargs={},
1439
+ lr_scheduler_type=SchedulerType.LINEAR,
1440
+ max_grad_norm=1.0,
1441
+ max_steps=-1,
1442
+ metric_for_best_model=loss,
1443
+ mp_parameters=,
1444
+ neftune_noise_alpha=None,
1445
+ no_cuda=False,
1446
+ num_train_epochs=20.0,
1447
+ optim=OptimizerNames.ADAMW_TORCH,
1448
+ optim_args=None,
1449
+ optim_target_modules=None,
1450
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1451
+ overwrite_output_dir=False,
1452
+ past_index=-1,
1453
+ per_device_eval_batch_size=8,
1454
+ per_device_train_batch_size=8,
1455
+ prediction_loss_only=False,
1456
+ push_to_hub=True,
1457
+ push_to_hub_model_id=None,
1458
+ push_to_hub_organization=None,
1459
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1460
+ ray_scope=last,
1461
+ remove_unused_columns=True,
1462
+ report_to=[],
1463
+ restore_callback_states_from_checkpoint=False,
1464
+ resume_from_checkpoint=None,
1465
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1466
+ save_on_each_node=False,
1467
+ save_only_model=False,
1468
+ save_safetensors=True,
1469
+ save_steps=500,
1470
+ save_strategy=IntervalStrategy.EPOCH,
1471
+ save_total_limit=None,
1472
+ seed=42,
1473
+ skip_memory_metrics=True,
1474
+ split_batches=None,
1475
+ tf32=None,
1476
+ torch_compile=False,
1477
+ torch_compile_backend=None,
1478
+ torch_compile_mode=None,
1479
+ torch_empty_cache_steps=None,
1480
+ torchdynamo=None,
1481
+ tpu_metrics_debug=False,
1482
+ tpu_num_cores=None,
1483
+ use_cpu=False,
1484
+ use_ipex=False,
1485
+ use_legacy_prediction_loop=False,
1486
+ use_mps_device=False,
1487
+ warmup_ratio=0.0,
1488
+ warmup_steps=500,
1489
+ weight_decay=0.0,
1490
+ )
1491
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1492
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1493
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1494
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1495
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1496
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1497
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
1498
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
1499
+ _n_gpu=1,
1500
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
1501
+ adafactor=False,
1502
+ adam_beta1=0.9,
1503
+ adam_beta2=0.999,
1504
+ adam_epsilon=1e-08,
1505
+ auto_find_batch_size=False,
1506
+ batch_eval_metrics=False,
1507
+ bf16=False,
1508
+ bf16_full_eval=False,
1509
+ data_seed=None,
1510
+ dataloader_drop_last=False,
1511
+ dataloader_num_workers=0,
1512
+ dataloader_persistent_workers=False,
1513
+ dataloader_pin_memory=True,
1514
+ dataloader_prefetch_factor=None,
1515
+ ddp_backend=None,
1516
+ ddp_broadcast_buffers=None,
1517
+ ddp_bucket_cap_mb=None,
1518
+ ddp_find_unused_parameters=None,
1519
+ ddp_timeout=1800,
1520
+ debug=[],
1521
+ deepspeed=None,
1522
+ disable_tqdm=False,
1523
+ dispatch_batches=None,
1524
+ do_eval=True,
1525
+ do_predict=False,
1526
+ do_train=True,
1527
+ eval_accumulation_steps=None,
1528
+ eval_delay=0,
1529
+ eval_do_concat_batches=True,
1530
+ eval_on_start=False,
1531
+ eval_steps=None,
1532
+ eval_strategy=IntervalStrategy.EPOCH,
1533
+ eval_use_gather_object=False,
1534
+ evaluation_strategy=epoch,
1535
+ fp16=False,
1536
+ fp16_backend=auto,
1537
+ fp16_full_eval=False,
1538
+ fp16_opt_level=O1,
1539
+ fsdp=[],
1540
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
1541
+ fsdp_min_num_params=0,
1542
+ fsdp_transformer_layer_cls_to_wrap=None,
1543
+ full_determinism=False,
1544
+ gradient_accumulation_steps=1,
1545
+ gradient_checkpointing=False,
1546
+ gradient_checkpointing_kwargs=None,
1547
+ greater_is_better=False,
1548
+ group_by_length=False,
1549
+ half_precision_backend=auto,
1550
+ hub_always_push=False,
1551
+ hub_model_id=None,
1552
+ hub_private_repo=False,
1553
+ hub_strategy=HubStrategy.EVERY_SAVE,
1554
+ hub_token=<HUB_TOKEN>,
1555
+ ignore_data_skip=False,
1556
+ include_inputs_for_metrics=False,
1557
+ include_num_input_tokens_seen=False,
1558
+ include_tokens_per_second=False,
1559
+ jit_mode_eval=False,
1560
+ label_names=None,
1561
+ label_smoothing_factor=0.0,
1562
+ learning_rate=5e-05,
1563
+ length_column_name=length,
1564
+ load_best_model_at_end=True,
1565
+ local_rank=0,
1566
+ log_level=passive,
1567
+ log_level_replica=warning,
1568
+ log_on_each_node=True,
1569
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-26-26_lmgpu-node-07,
1570
+ logging_first_step=False,
1571
+ logging_nan_inf_filter=True,
1572
+ logging_steps=500,
1573
+ logging_strategy=IntervalStrategy.EPOCH,
1574
+ lr_scheduler_kwargs={},
1575
+ lr_scheduler_type=SchedulerType.LINEAR,
1576
+ max_grad_norm=1.0,
1577
+ max_steps=-1,
1578
+ metric_for_best_model=loss,
1579
+ mp_parameters=,
1580
+ neftune_noise_alpha=None,
1581
+ no_cuda=False,
1582
+ num_train_epochs=20.0,
1583
+ optim=OptimizerNames.ADAMW_TORCH,
1584
+ optim_args=None,
1585
+ optim_target_modules=None,
1586
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1587
+ overwrite_output_dir=False,
1588
+ past_index=-1,
1589
+ per_device_eval_batch_size=8,
1590
+ per_device_train_batch_size=8,
1591
+ prediction_loss_only=False,
1592
+ push_to_hub=True,
1593
+ push_to_hub_model_id=None,
1594
+ push_to_hub_organization=None,
1595
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1596
+ ray_scope=last,
1597
+ remove_unused_columns=True,
1598
+ report_to=[],
1599
+ restore_callback_states_from_checkpoint=False,
1600
+ resume_from_checkpoint=None,
1601
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1602
+ save_on_each_node=False,
1603
+ save_only_model=False,
1604
+ save_safetensors=True,
1605
+ save_steps=500,
1606
+ save_strategy=IntervalStrategy.EPOCH,
1607
+ save_total_limit=None,
1608
+ seed=42,
1609
+ skip_memory_metrics=True,
1610
+ split_batches=None,
1611
+ tf32=None,
1612
+ torch_compile=False,
1613
+ torch_compile_backend=None,
1614
+ torch_compile_mode=None,
1615
+ torch_empty_cache_steps=None,
1616
+ torchdynamo=None,
1617
+ tpu_metrics_debug=False,
1618
+ tpu_num_cores=None,
1619
+ use_cpu=False,
1620
+ use_ipex=False,
1621
+ use_legacy_prediction_loop=False,
1622
+ use_mps_device=False,
1623
+ warmup_ratio=0.0,
1624
+ warmup_steps=500,
1625
+ weight_decay=0.0,
1626
+ )
1627
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1628
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1629
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1630
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1631
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1632
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1633
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
1634
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
1635
+ _n_gpu=1,
1636
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
1637
+ adafactor=False,
1638
+ adam_beta1=0.9,
1639
+ adam_beta2=0.999,
1640
+ adam_epsilon=1e-08,
1641
+ auto_find_batch_size=False,
1642
+ batch_eval_metrics=False,
1643
+ bf16=False,
1644
+ bf16_full_eval=False,
1645
+ data_seed=None,
1646
+ dataloader_drop_last=False,
1647
+ dataloader_num_workers=0,
1648
+ dataloader_persistent_workers=False,
1649
+ dataloader_pin_memory=True,
1650
+ dataloader_prefetch_factor=None,
1651
+ ddp_backend=None,
1652
+ ddp_broadcast_buffers=None,
1653
+ ddp_bucket_cap_mb=None,
1654
+ ddp_find_unused_parameters=None,
1655
+ ddp_timeout=1800,
1656
+ debug=[],
1657
+ deepspeed=None,
1658
+ disable_tqdm=False,
1659
+ dispatch_batches=None,
1660
+ do_eval=True,
1661
+ do_predict=False,
1662
+ do_train=True,
1663
+ eval_accumulation_steps=None,
1664
+ eval_delay=0,
1665
+ eval_do_concat_batches=True,
1666
+ eval_on_start=False,
1667
+ eval_steps=None,
1668
+ eval_strategy=IntervalStrategy.EPOCH,
1669
+ eval_use_gather_object=False,
1670
+ evaluation_strategy=epoch,
1671
+ fp16=False,
1672
+ fp16_backend=auto,
1673
+ fp16_full_eval=False,
1674
+ fp16_opt_level=O1,
1675
+ fsdp=[],
1676
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
1677
+ fsdp_min_num_params=0,
1678
+ fsdp_transformer_layer_cls_to_wrap=None,
1679
+ full_determinism=False,
1680
+ gradient_accumulation_steps=1,
1681
+ gradient_checkpointing=False,
1682
+ gradient_checkpointing_kwargs=None,
1683
+ greater_is_better=False,
1684
+ group_by_length=False,
1685
+ half_precision_backend=auto,
1686
+ hub_always_push=False,
1687
+ hub_model_id=None,
1688
+ hub_private_repo=False,
1689
+ hub_strategy=HubStrategy.EVERY_SAVE,
1690
+ hub_token=<HUB_TOKEN>,
1691
+ ignore_data_skip=False,
1692
+ include_inputs_for_metrics=False,
1693
+ include_num_input_tokens_seen=False,
1694
+ include_tokens_per_second=False,
1695
+ jit_mode_eval=False,
1696
+ label_names=None,
1697
+ label_smoothing_factor=0.0,
1698
+ learning_rate=5e-05,
1699
+ length_column_name=length,
1700
+ load_best_model_at_end=True,
1701
+ local_rank=0,
1702
+ log_level=passive,
1703
+ log_level_replica=warning,
1704
+ log_on_each_node=True,
1705
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_20-31-56_lmgpu-node-07,
1706
+ logging_first_step=False,
1707
+ logging_nan_inf_filter=True,
1708
+ logging_steps=500,
1709
+ logging_strategy=IntervalStrategy.EPOCH,
1710
+ lr_scheduler_kwargs={},
1711
+ lr_scheduler_type=SchedulerType.LINEAR,
1712
+ max_grad_norm=1.0,
1713
+ max_steps=-1,
1714
+ metric_for_best_model=loss,
1715
+ mp_parameters=,
1716
+ neftune_noise_alpha=None,
1717
+ no_cuda=False,
1718
+ num_train_epochs=20.0,
1719
+ optim=OptimizerNames.ADAMW_TORCH,
1720
+ optim_args=None,
1721
+ optim_target_modules=None,
1722
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1723
+ overwrite_output_dir=False,
1724
+ past_index=-1,
1725
+ per_device_eval_batch_size=8,
1726
+ per_device_train_batch_size=8,
1727
+ prediction_loss_only=False,
1728
+ push_to_hub=True,
1729
+ push_to_hub_model_id=None,
1730
+ push_to_hub_organization=None,
1731
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1732
+ ray_scope=last,
1733
+ remove_unused_columns=True,
1734
+ report_to=[],
1735
+ restore_callback_states_from_checkpoint=False,
1736
+ resume_from_checkpoint=None,
1737
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1738
+ save_on_each_node=False,
1739
+ save_only_model=False,
1740
+ save_safetensors=True,
1741
+ save_steps=500,
1742
+ save_strategy=IntervalStrategy.EPOCH,
1743
+ save_total_limit=None,
1744
+ seed=42,
1745
+ skip_memory_metrics=True,
1746
+ split_batches=None,
1747
+ tf32=None,
1748
+ torch_compile=False,
1749
+ torch_compile_backend=None,
1750
+ torch_compile_mode=None,
1751
+ torch_empty_cache_steps=None,
1752
+ torchdynamo=None,
1753
+ tpu_metrics_debug=False,
1754
+ tpu_num_cores=None,
1755
+ use_cpu=False,
1756
+ use_ipex=False,
1757
+ use_legacy_prediction_loop=False,
1758
+ use_mps_device=False,
1759
+ warmup_ratio=0.0,
1760
+ warmup_steps=500,
1761
+ weight_decay=0.0,
1762
+ )
1763
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1764
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1765
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1766
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1767
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1768
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1769
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-f1f2b177e1af8283.arrow
1770
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-b679eeba267acbff.arrow
1771
+ WARNING:__main__:The tokenizer picked seems to have a very large `model_max_length` (1000000000000000019884624838656). Using block_size=1024 instead. You can change that default value by passing --block_size xxx.
1772
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-d68c2e438392fa5c.arrow
1773
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-0e72b5f0f52176a9.arrow
1774
+ WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
1775
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
1776
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
1777
+ _n_gpu=1,
1778
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
1779
+ adafactor=False,
1780
+ adam_beta1=0.9,
1781
+ adam_beta2=0.999,
1782
+ adam_epsilon=1e-08,
1783
+ auto_find_batch_size=False,
1784
+ batch_eval_metrics=False,
1785
+ bf16=False,
1786
+ bf16_full_eval=False,
1787
+ data_seed=None,
1788
+ dataloader_drop_last=False,
1789
+ dataloader_num_workers=0,
1790
+ dataloader_persistent_workers=False,
1791
+ dataloader_pin_memory=True,
1792
+ dataloader_prefetch_factor=None,
1793
+ ddp_backend=None,
1794
+ ddp_broadcast_buffers=None,
1795
+ ddp_bucket_cap_mb=None,
1796
+ ddp_find_unused_parameters=None,
1797
+ ddp_timeout=1800,
1798
+ debug=[],
1799
+ deepspeed=None,
1800
+ disable_tqdm=False,
1801
+ dispatch_batches=None,
1802
+ do_eval=True,
1803
+ do_predict=False,
1804
+ do_train=True,
1805
+ eval_accumulation_steps=None,
1806
+ eval_delay=0,
1807
+ eval_do_concat_batches=True,
1808
+ eval_on_start=False,
1809
+ eval_steps=None,
1810
+ eval_strategy=IntervalStrategy.EPOCH,
1811
+ eval_use_gather_object=False,
1812
+ evaluation_strategy=epoch,
1813
+ fp16=False,
1814
+ fp16_backend=auto,
1815
+ fp16_full_eval=False,
1816
+ fp16_opt_level=O1,
1817
+ fsdp=[],
1818
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
1819
+ fsdp_min_num_params=0,
1820
+ fsdp_transformer_layer_cls_to_wrap=None,
1821
+ full_determinism=False,
1822
+ gradient_accumulation_steps=1,
1823
+ gradient_checkpointing=False,
1824
+ gradient_checkpointing_kwargs=None,
1825
+ greater_is_better=False,
1826
+ group_by_length=False,
1827
+ half_precision_backend=auto,
1828
+ hub_always_push=False,
1829
+ hub_model_id=None,
1830
+ hub_private_repo=False,
1831
+ hub_strategy=HubStrategy.EVERY_SAVE,
1832
+ hub_token=<HUB_TOKEN>,
1833
+ ignore_data_skip=False,
1834
+ include_inputs_for_metrics=False,
1835
+ include_num_input_tokens_seen=False,
1836
+ include_tokens_per_second=False,
1837
+ jit_mode_eval=False,
1838
+ label_names=None,
1839
+ label_smoothing_factor=0.0,
1840
+ learning_rate=5e-05,
1841
+ length_column_name=length,
1842
+ load_best_model_at_end=True,
1843
+ local_rank=0,
1844
+ log_level=passive,
1845
+ log_level_replica=warning,
1846
+ log_on_each_node=True,
1847
+ logging_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base/runs/Sep07_21-03-25_lmgpu-node-07,
1848
+ logging_first_step=False,
1849
+ logging_nan_inf_filter=True,
1850
+ logging_steps=500,
1851
+ logging_strategy=IntervalStrategy.EPOCH,
1852
+ lr_scheduler_kwargs={},
1853
+ lr_scheduler_type=SchedulerType.LINEAR,
1854
+ max_grad_norm=1.0,
1855
+ max_steps=-1,
1856
+ metric_for_best_model=loss,
1857
+ mp_parameters=,
1858
+ neftune_noise_alpha=None,
1859
+ no_cuda=False,
1860
+ num_train_epochs=20.0,
1861
+ optim=OptimizerNames.ADAMW_TORCH,
1862
+ optim_args=None,
1863
+ optim_target_modules=None,
1864
+ output_dir=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1865
+ overwrite_output_dir=False,
1866
+ past_index=-1,
1867
+ per_device_eval_batch_size=8,
1868
+ per_device_train_batch_size=8,
1869
+ prediction_loss_only=False,
1870
+ push_to_hub=True,
1871
+ push_to_hub_model_id=None,
1872
+ push_to_hub_organization=None,
1873
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
1874
+ ray_scope=last,
1875
+ remove_unused_columns=True,
1876
+ report_to=[],
1877
+ restore_callback_states_from_checkpoint=False,
1878
+ resume_from_checkpoint=None,
1879
+ run_name=/home/iais_marenpielka/Bouthaina/res_nw_eg_aragpt2-base,
1880
+ save_on_each_node=False,
1881
+ save_only_model=False,
1882
+ save_safetensors=True,
1883
+ save_steps=500,
1884
+ save_strategy=IntervalStrategy.EPOCH,
1885
+ save_total_limit=None,
1886
+ seed=42,
1887
+ skip_memory_metrics=True,
1888
+ split_batches=None,
1889
+ tf32=None,
1890
+ torch_compile=False,
1891
+ torch_compile_backend=None,
1892
+ torch_compile_mode=None,
1893
+ torch_empty_cache_steps=None,
1894
+ torchdynamo=None,
1895
+ tpu_metrics_debug=False,
1896
+ tpu_num_cores=None,
1897
+ use_cpu=False,
1898
+ use_ipex=False,
1899
+ use_legacy_prediction_loop=False,
1900
+ use_mps_device=False,
1901
+ warmup_ratio=0.0,
1902
+ warmup_steps=500,
1903
+ weight_decay=0.0,
1904
+ )
1905
+ INFO:datasets.builder:Using custom data configuration default-8c97581fc2299c6f
1906
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
1907
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
1908
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1909
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
1910
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
1911
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-f1f2b177e1af8283.arrow
1912
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-ec15ec265c0b61ef.arrow
1913
+ WARNING:__main__:The tokenizer picked seems to have a very large `model_max_length` (1000000000000000019884624838656). Using block_size=1024 instead. You can change that default value by passing --block_size xxx.
1914
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-d68c2e438392fa5c.arrow
1915
+ INFO:datasets.arrow_dataset:Caching processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-8c97581fc2299c6f/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-427224d0e380c92b.arrow
1916
+ WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
1917
+ INFO:root:Epoch 1.0: Train Loss = None, Eval Loss = None
1918
+ INFO:absl:Using default tokenizer.
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2fb2f3b8963636220c1ce9eafe2a474c36cc0bbb94dc2789abb59729eab5a7c6
3
+ size 540004992
special_tokens_map.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ {
4
+ "content": "<sep>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false
9
+ }
10
+ ],
11
+ "bos_token": {
12
+ "content": "<|bos|>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false
17
+ },
18
+ "eos_token": {
19
+ "content": "<|endoftext|>",
20
+ "lstrip": false,
21
+ "normalized": false,
22
+ "rstrip": false,
23
+ "single_word": false
24
+ },
25
+ "pad_token": {
26
+ "content": "[PAD]",
27
+ "lstrip": false,
28
+ "normalized": false,
29
+ "rstrip": false,
30
+ "single_word": false
31
+ },
32
+ "unk_token": {
33
+ "content": "<|unk|>",
34
+ "lstrip": false,
35
+ "normalized": false,
36
+ "rstrip": false,
37
+ "single_word": false
38
+ }
39
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "0": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "1": {
13
+ "content": "<s>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "2": {
21
+ "content": "<pad>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "3": {
29
+ "content": "</s>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "64000": {
37
+ "content": "<|bos|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "64001": {
45
+ "content": "<|unk|>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "64002": {
53
+ "content": "[PAD]",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "64003": {
61
+ "content": "<sep>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ }
68
+ },
69
+ "additional_special_tokens": [
70
+ "<sep>"
71
+ ],
72
+ "bos_token": "<|bos|>",
73
+ "clean_up_tokenization_spaces": true,
74
+ "eos_token": "<|endoftext|>",
75
+ "model_max_length": 1000000000000000019884624838656,
76
+ "pad_token": "[PAD]",
77
+ "tokenizer_class": "GPT2Tokenizer",
78
+ "unk_token": "<|unk|>"
79
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c4180db3ae9ec11d4f2583bfcf1e7e78403369f2fabe1bf4ffbe984ff13fe598
3
+ size 5240
vocab.json ADDED
The diff for this file is too large to render. See raw diff