lapp0 commited on
Commit
4b2f194
1 Parent(s): bd05b9a

End of training

Browse files
README.md CHANGED
@@ -75,7 +75,7 @@ More information needed
75
  <br/>
76
 
77
  # Train Dataset
78
- Trained on 521,346,447 tokens from the [wikimedia/wikipedia](https://huggingface.co/datasets/wikimedia/wikipedia) dataset.
79
 
80
  - Num Samples: `990,000`
81
  - Subset: `20231101.en`
@@ -94,7 +94,7 @@ The following hyperparameters were used during training:
94
  <details>
95
  <summary>Expand</summary>
96
 
97
- - learning_rate: `0.0002`
98
  - train_batch_size: `16`
99
  - eval_batch_size: `8`
100
  - seed: `42`
@@ -103,7 +103,7 @@ The following hyperparameters were used during training:
103
  - num_epochs: `1.0`
104
  - distillation_objective: `DistillationObjective(logits_loss_component=LossComponent(label=logits, weight=1, loss_fn=kl), attn_loss_component=LossComponent(label=attn, weight=5, loss_fn=raw_mse, layer_mapper=layer-2, norm=layernorm_teacher_only, projector=orthogonal))`
105
  - train_embeddings: `True`
106
- - lr_scheduler: `<torch.optim.lr_scheduler.LambdaLR object at 0x7fc354bdc1c0>`
107
  - student_model_name_or_path: `None`
108
  - student_config_name_or_path: `distilbert/distilgpt2`
109
  - student_model_config: `None`
@@ -133,6 +133,6 @@ The following hyperparameters were used during training:
133
 
134
  # Framework Versions
135
  - Distily 0.4.1
136
- - Transformers 4.44.1
137
  - Pytorch 2.4.0+cu121
138
- - Datasets 2.21.0
 
75
  <br/>
76
 
77
  # Train Dataset
78
+ Trained on 521,327,470 tokens from the [wikimedia/wikipedia](https://huggingface.co/datasets/wikimedia/wikipedia) dataset.
79
 
80
  - Num Samples: `990,000`
81
  - Subset: `20231101.en`
 
94
  <details>
95
  <summary>Expand</summary>
96
 
97
+ - learning_rate: `0.0001`
98
  - train_batch_size: `16`
99
  - eval_batch_size: `8`
100
  - seed: `42`
 
103
  - num_epochs: `1.0`
104
  - distillation_objective: `DistillationObjective(logits_loss_component=LossComponent(label=logits, weight=1, loss_fn=kl), attn_loss_component=LossComponent(label=attn, weight=5, loss_fn=raw_mse, layer_mapper=layer-2, norm=layernorm_teacher_only, projector=orthogonal))`
105
  - train_embeddings: `True`
106
+ - lr_scheduler: `<torch.optim.lr_scheduler.LambdaLR object at 0x7f3be9424d60>`
107
  - student_model_name_or_path: `None`
108
  - student_config_name_or_path: `distilbert/distilgpt2`
109
  - student_model_config: `None`
 
133
 
134
  # Framework Versions
135
  - Distily 0.4.1
136
+ - Transformers 4.44.2
137
  - Pytorch 2.4.0+cu121
138
+ - Datasets 2.18.0
config.json CHANGED
@@ -40,7 +40,7 @@
40
  }
41
  },
42
  "torch_dtype": "bfloat16",
43
- "transformers_version": "4.44.1",
44
  "use_cache": true,
45
  "vocab_size": 50257
46
  }
 
40
  }
41
  },
42
  "torch_dtype": "bfloat16",
43
+ "transformers_version": "4.44.2",
44
  "use_cache": true,
45
  "vocab_size": 50257
46
  }
generation_config.json CHANGED
@@ -2,5 +2,5 @@
2
  "_from_model_config": true,
3
  "bos_token_id": 50256,
4
  "eos_token_id": 50256,
5
- "transformers_version": "4.44.1"
6
  }
 
2
  "_from_model_config": true,
3
  "bos_token_id": 50256,
4
  "eos_token_id": 50256,
5
+ "transformers_version": "4.44.2"
6
  }
logs/attn_norm=layernorm_teacher_only, attn_projector=orthogonal, attn_weight=5, learning_rate=0.0001, per_device_train_batch_size=16, warmup_ratio=0/events.out.tfevents.1725178640.a7e428977e35 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:204952719e1a148787309f60c4b1b19583893cc25486e40c17636e926c1a428d
3
+ size 29625562
logs/attn_norm=layernorm_teacher_only, attn_projector=orthogonal, attn_weight=5, learning_rate=0.0001, per_device_train_batch_size=16, warmup_ratio=0/events.out.tfevents.1725196913.a7e428977e35 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae11298ea99efd39b47ff1666d984deb85ad7ab90dee3e63a78e947a503910a6
3
+ size 529
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6318f9b2822e502bfafc3ad445305539c4863f2ecd2d7f62dc8f90dbee699c24
3
  size 163832792
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4f2ef28dde231485eb988fcef4eb1d4e4e0e2b1c45998e4d2e296930d85f65bb
3
  size 163832792
training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a963627fe44f0185f6f7c2a07eb6c2ae9ad612ad008a54151f48f4ea9ec70130
3
  size 5624
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e428b3aa75f06cdbd4f4769259e5966750e3cff19e4c302c76ce2d2a1e523af7
3
  size 5624