nlparabic commited on
Commit
772af4a
·
verified ·
1 Parent(s): 0c6bb21

Training in progress, step 319

Browse files
Files changed (3) hide show
  1. egy_training_log.txt +141 -4
  2. model.safetensors +1 -1
  3. training_args.bin +1 -1
egy_training_log.txt CHANGED
@@ -1,6 +1,143 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
2
  WARNING:root:Epoch 1.0: No losses recorded yet.
3
- INFO:absl:Using default tokenizer.
4
- INFO:root:Epoch 2.0: Train Loss = 4.3013, Eval Loss = 3.128293514251709
5
- INFO:__main__:*** Evaluate ***
6
- INFO:absl:Using default tokenizer.
 
1
+ WARNING:__main__:Process rank: 0, device: cuda:0, n_gpu: 1, distributed training: False, 16-bits training: False
2
+ INFO:__main__:Training/evaluation parameters TrainingArguments(
3
+ _n_gpu=1,
4
+ accelerator_config={'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None, 'use_configured_state': False},
5
+ adafactor=False,
6
+ adam_beta1=0.9,
7
+ adam_beta2=0.999,
8
+ adam_epsilon=1e-08,
9
+ auto_find_batch_size=False,
10
+ batch_eval_metrics=False,
11
+ bf16=False,
12
+ bf16_full_eval=False,
13
+ data_seed=None,
14
+ dataloader_drop_last=False,
15
+ dataloader_num_workers=0,
16
+ dataloader_persistent_workers=False,
17
+ dataloader_pin_memory=True,
18
+ dataloader_prefetch_factor=None,
19
+ ddp_backend=None,
20
+ ddp_broadcast_buffers=None,
21
+ ddp_bucket_cap_mb=None,
22
+ ddp_find_unused_parameters=None,
23
+ ddp_timeout=1800,
24
+ debug=[],
25
+ deepspeed=None,
26
+ disable_tqdm=False,
27
+ dispatch_batches=None,
28
+ do_eval=True,
29
+ do_predict=False,
30
+ do_train=True,
31
+ eval_accumulation_steps=None,
32
+ eval_delay=0,
33
+ eval_do_concat_batches=True,
34
+ eval_on_start=False,
35
+ eval_steps=500,
36
+ eval_strategy=IntervalStrategy.STEPS,
37
+ eval_use_gather_object=False,
38
+ evaluation_strategy=steps,
39
+ fp16=False,
40
+ fp16_backend=auto,
41
+ fp16_full_eval=False,
42
+ fp16_opt_level=O1,
43
+ fsdp=[],
44
+ fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False},
45
+ fsdp_min_num_params=0,
46
+ fsdp_transformer_layer_cls_to_wrap=None,
47
+ full_determinism=False,
48
+ gradient_accumulation_steps=1,
49
+ gradient_checkpointing=False,
50
+ gradient_checkpointing_kwargs=None,
51
+ greater_is_better=False,
52
+ group_by_length=False,
53
+ half_precision_backend=auto,
54
+ hub_always_push=False,
55
+ hub_model_id=None,
56
+ hub_private_repo=False,
57
+ hub_strategy=HubStrategy.EVERY_SAVE,
58
+ hub_token=<HUB_TOKEN>,
59
+ ignore_data_skip=False,
60
+ include_inputs_for_metrics=False,
61
+ include_num_input_tokens_seen=False,
62
+ include_tokens_per_second=False,
63
+ jit_mode_eval=False,
64
+ label_names=None,
65
+ label_smoothing_factor=0.0,
66
+ learning_rate=5e-05,
67
+ length_column_name=length,
68
+ load_best_model_at_end=True,
69
+ local_rank=0,
70
+ log_level=passive,
71
+ log_level_replica=warning,
72
+ log_on_each_node=True,
73
+ logging_dir=/tmp/test-egy_aragpt/runs/Aug25_11-41-29_lmgpu-node-09,
74
+ logging_first_step=False,
75
+ logging_nan_inf_filter=True,
76
+ logging_steps=500,
77
+ logging_strategy=IntervalStrategy.STEPS,
78
+ lr_scheduler_kwargs={},
79
+ lr_scheduler_type=SchedulerType.LINEAR,
80
+ max_grad_norm=1.0,
81
+ max_steps=-1,
82
+ metric_for_best_model=loss,
83
+ mp_parameters=,
84
+ neftune_noise_alpha=None,
85
+ no_cuda=False,
86
+ num_train_epochs=1.0,
87
+ optim=OptimizerNames.ADAMW_TORCH,
88
+ optim_args=None,
89
+ optim_target_modules=None,
90
+ output_dir=/tmp/test-egy_aragpt,
91
+ overwrite_output_dir=False,
92
+ past_index=-1,
93
+ per_device_eval_batch_size=8,
94
+ per_device_train_batch_size=8,
95
+ prediction_loss_only=False,
96
+ push_to_hub=True,
97
+ push_to_hub_model_id=None,
98
+ push_to_hub_organization=None,
99
+ push_to_hub_token=<PUSH_TO_HUB_TOKEN>,
100
+ ray_scope=last,
101
+ remove_unused_columns=True,
102
+ report_to=[],
103
+ restore_callback_states_from_checkpoint=False,
104
+ resume_from_checkpoint=None,
105
+ run_name=/tmp/test-egy_aragpt,
106
+ save_on_each_node=False,
107
+ save_only_model=False,
108
+ save_safetensors=True,
109
+ save_steps=500,
110
+ save_strategy=IntervalStrategy.STEPS,
111
+ save_total_limit=None,
112
+ seed=42,
113
+ skip_memory_metrics=True,
114
+ split_batches=None,
115
+ tf32=None,
116
+ torch_compile=False,
117
+ torch_compile_backend=None,
118
+ torch_compile_mode=None,
119
+ torch_empty_cache_steps=None,
120
+ torchdynamo=None,
121
+ tpu_metrics_debug=False,
122
+ tpu_num_cores=None,
123
+ use_cpu=False,
124
+ use_ipex=False,
125
+ use_legacy_prediction_loop=False,
126
+ use_mps_device=False,
127
+ warmup_ratio=0.0,
128
+ warmup_steps=500,
129
+ weight_decay=0.0,
130
+ )
131
+ INFO:datasets.builder:Using custom data configuration default-0637777c38512acf
132
+ INFO:datasets.info:Loading Dataset Infos from /home/iais_marenpielka/Bouthaina/miniconda3/lib/python3.12/site-packages/datasets/packaged_modules/text
133
+ INFO:datasets.builder:Overwrite dataset info from restored data version if exists.
134
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
135
+ INFO:datasets.builder:Found cached dataset text (/home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101)
136
+ INFO:datasets.info:Loading Dataset info from /home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101
137
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-37ea47da1c3ae86a.arrow
138
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-d1d890b04a73f183.arrow
139
+ WARNING:__main__:The tokenizer picked seems to have a very large `model_max_length` (1000000000000000019884624838656). Using block_size=768 instead. You can change that default value by passing --block_size xxx.
140
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-607ae57e4b4160b3.arrow
141
+ INFO:datasets.arrow_dataset:Loading cached processed dataset at /home/iais_marenpielka/.cache/huggingface/datasets/text/default-0637777c38512acf/0.0.0/96636a050ef51804b84abbfd4f4ad440e01153c24b86293eb5c3b300a41f9101/cache-4ddbb6e08bb37d3f.arrow
142
  WARNING:accelerate.utils.other:Detected kernel version 5.4.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
143
  WARNING:root:Epoch 1.0: No losses recorded yet.
 
 
 
 
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a016bd2c2582c90275dbe0b2997c0b9983c070dea72ed95a1844ed2a68dc25aa
3
  size 539218560
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e68681566aeb957c233b5b78153c8c29a8344167a2e86d8636e264bf8897ca19
3
  size 539218560
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c5864db0cf974d938ed7888414270192f1cfbc55855c25f1607a6df6aa44269e
3
  size 5176
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1a40a2e008b46bca9cc778715d72bfaf171d9b47489e04239edc2db1541a42ae
3
  size 5176