Training in progress epoch 0

Browse files

Files changed (4) hide show

README.md +8 -17
config.json +5 -5
tf_model.h5 +2 -2
tokenizer_config.json +1 -1

README.md CHANGED Viewed

@@ -1,6 +1,6 @@
 ---
 license: apache-2.0
-base_model: google/electra-base-generator
 tags:
 - generated_from_keras_callback
 model-index:
@@ -13,11 +13,11 @@ probably proofread and complete it, then remove this comment. -->
 # syabusyabu0141/test
-This model is a fine-tuned version of [google/electra-base-generator](https://huggingface.co/google/electra-base-generator) on an unknown dataset.
 It achieves the following results on the evaluation set:
-- Train Loss: 0.4571
-- Validation Loss: 1.1024
-- Epoch: 9
 ## Model description
@@ -36,23 +36,14 @@ More information needed
 ### Training hyperparameters
 The following hyperparameters were used during training:
-- optimizer: {'name': 'AdamWeightDecay', 'learning_rate': 1e-05, 'decay': 0.0, 'beta_1': 0.9, 'beta_2': 0.999, 'epsilon': 1e-07, 'amsgrad': False, 'weight_decay_rate': 0.01}
 - training_precision: float32
 ### Training results
 | Train Loss | Validation Loss | Epoch |
 |:----------:|:---------------:|:-----:|
-| 1.3171     | 1.1419          | 0     |
-| 1.1021     | 1.0832          | 1     |
-| 0.9797     | 1.0280          | 2     |
-| 0.8696     | 1.0097          | 3     |
-| 0.7756     | 0.9988          | 4     |
-| 0.6957     | 0.9989          | 5     |
-| 0.6246     | 1.0204          | 6     |
-| 0.5659     | 1.0395          | 7     |
-| 0.5053     | 1.0549          | 8     |
-| 0.4571     | 1.1024          | 9     |
 ### Framework versions
@@ -60,4 +51,4 @@ The following hyperparameters were used during training:
 - Transformers 4.34.0
 - TensorFlow 2.13.0
 - Datasets 2.14.5
-- Tokenizers 0.14.0

 ---
 license: apache-2.0
+base_model: google/electra-base-discriminator
 tags:
 - generated_from_keras_callback
 model-index:
 # syabusyabu0141/test
+This model is a fine-tuned version of [google/electra-base-discriminator](https://huggingface.co/google/electra-base-discriminator) on an unknown dataset.
 It achieves the following results on the evaluation set:
+- Train Loss: 0.2559
+- Validation Loss: 0.1547
+- Epoch: 0
 ## Model description
 ### Training hyperparameters
 The following hyperparameters were used during training:
+- optimizer: {'name': 'Adam', 'weight_decay': None, 'clipnorm': None, 'global_clipnorm': None, 'clipvalue': None, 'use_ema': False, 'ema_momentum': 0.99, 'ema_overwrite_frequency': None, 'jit_compile': True, 'is_legacy_optimizer': False, 'learning_rate': {'module': 'keras.optimizers.schedules', 'class_name': 'PolynomialDecay', 'config': {'initial_learning_rate': 1e-05, 'decay_steps': 6924, 'end_learning_rate': 0.0, 'power': 1.0, 'cycle': False, 'name': None}, 'registered_name': None}, 'beta_1': 0.9, 'beta_2': 0.999, 'epsilon': 1e-08, 'amsgrad': False}
 - training_precision: float32
 ### Training results
 | Train Loss | Validation Loss | Epoch |
 |:----------:|:---------------:|:-----:|
+| 0.2559     | 0.1547          | 0     |
 ### Framework versions
 - Transformers 4.34.0
 - TensorFlow 2.13.0
 - Datasets 2.14.5
+- Tokenizers 0.14.1

config.json CHANGED Viewed

@@ -1,20 +1,20 @@
 {
-  "_name_or_path": "google/electra-base-generator",
   "architectures": [
-    "ElectraForMaskedLM"
   ],
   "attention_probs_dropout_prob": 0.1,
   "classifier_dropout": null,
   "embedding_size": 768,
   "hidden_act": "gelu",
   "hidden_dropout_prob": 0.1,
-  "hidden_size": 256,
   "initializer_range": 0.02,
-  "intermediate_size": 1024,
   "layer_norm_eps": 1e-12,
   "max_position_embeddings": 512,
   "model_type": "electra",
-  "num_attention_heads": 4,
   "num_hidden_layers": 12,
   "pad_token_id": 0,
   "position_embedding_type": "absolute",

 {
+  "_name_or_path": "google/electra-base-discriminator",
   "architectures": [
+    "ElectraForSequenceClassification"
   ],
   "attention_probs_dropout_prob": 0.1,
   "classifier_dropout": null,
   "embedding_size": 768,
   "hidden_act": "gelu",
   "hidden_dropout_prob": 0.1,
+  "hidden_size": 768,
   "initializer_range": 0.02,
+  "intermediate_size": 3072,
   "layer_norm_eps": 1e-12,
   "max_position_embeddings": 512,
   "model_type": "electra",
+  "num_attention_heads": 12,
   "num_hidden_layers": 12,
   "pad_token_id": 0,
   "position_embedding_type": "absolute",

tf_model.h5 CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:431991482959055621c43b55bb023b17e5a9b8f5617c1d7b13a86551e9182f6b
-size 230595504

 version https://git-lfs.github.com/spec/v1
+oid sha256:a118e198a060071902e0dc717ba6dbe25e41f4955a301121ffc68f80f8074ee8
+size 438225104

tokenizer_config.json CHANGED Viewed

@@ -54,6 +54,6 @@
   "strip_accents": null,
   "tokenize_chinese_chars": true,
   "tokenizer_class": "ElectraTokenizer",
-  "tokenizer_file": "/root/.cache/huggingface/hub/models--google--electra-base-generator/snapshots/1c65e3f5f4597679b87620707df5774c08c6606d/tokenizer.json",
   "unk_token": "[UNK]"
 }

   "strip_accents": null,
   "tokenize_chinese_chars": true,
   "tokenizer_class": "ElectraTokenizer",
+  "tokenizer_file": "/root/.cache/huggingface/hub/models--google--electra-base-discriminator/snapshots/1b48ef100dac4676d84125a8a7b7ab7c51e00386/tokenizer.json",
   "unk_token": "[UNK]"
 }