cvapict commited on 6 days ago

Commit

f52379d

•

1 Parent(s): 4d497db

End of training

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

README.md +70 -0
config.json +26 -0
model.safetensors +3 -0
run-0i7sidmz/checkpoint-100/config.json +26 -0
run-0i7sidmz/checkpoint-100/model.safetensors +3 -0
run-0i7sidmz/checkpoint-100/optimizer.pt +3 -0
run-0i7sidmz/checkpoint-100/rng_state.pth +3 -0
run-0i7sidmz/checkpoint-100/scheduler.pt +3 -0
run-0i7sidmz/checkpoint-100/special_tokens_map.json +7 -0
run-0i7sidmz/checkpoint-100/tokenizer.json +0 -0
run-0i7sidmz/checkpoint-100/tokenizer_config.json +55 -0
run-0i7sidmz/checkpoint-100/trainer_state.json +123 -0
run-0i7sidmz/checkpoint-100/training_args.bin +3 -0
run-0i7sidmz/checkpoint-100/vocab.txt +0 -0
run-0nyui8bu/checkpoint-25/config.json +26 -0
run-0nyui8bu/checkpoint-25/model.safetensors +3 -0
run-0nyui8bu/checkpoint-25/optimizer.pt +3 -0
run-0nyui8bu/checkpoint-25/rng_state.pth +3 -0
run-0nyui8bu/checkpoint-25/scheduler.pt +3 -0
run-0nyui8bu/checkpoint-25/special_tokens_map.json +7 -0
run-0nyui8bu/checkpoint-25/tokenizer.json +0 -0
run-0nyui8bu/checkpoint-25/tokenizer_config.json +55 -0
run-0nyui8bu/checkpoint-25/trainer_state.json +67 -0
run-0nyui8bu/checkpoint-25/training_args.bin +3 -0
run-0nyui8bu/checkpoint-25/vocab.txt +0 -0
run-4r5wbvc3/checkpoint-50/config.json +26 -0
run-4r5wbvc3/checkpoint-50/model.safetensors +3 -0
run-4r5wbvc3/checkpoint-50/optimizer.pt +3 -0
run-4r5wbvc3/checkpoint-50/rng_state.pth +3 -0
run-4r5wbvc3/checkpoint-50/scheduler.pt +3 -0
run-4r5wbvc3/checkpoint-50/special_tokens_map.json +7 -0
run-4r5wbvc3/checkpoint-50/tokenizer.json +0 -0
run-4r5wbvc3/checkpoint-50/tokenizer_config.json +55 -0
run-4r5wbvc3/checkpoint-50/trainer_state.json +88 -0
run-4r5wbvc3/checkpoint-50/training_args.bin +3 -0
run-4r5wbvc3/checkpoint-50/vocab.txt +0 -0
run-7dukmcwd/checkpoint-400/config.json +26 -0
run-7dukmcwd/checkpoint-400/model.safetensors +3 -0
run-7dukmcwd/checkpoint-400/optimizer.pt +3 -0
run-7dukmcwd/checkpoint-400/rng_state.pth +3 -0
run-7dukmcwd/checkpoint-400/scheduler.pt +3 -0
run-7dukmcwd/checkpoint-400/special_tokens_map.json +7 -0
run-7dukmcwd/checkpoint-400/tokenizer.json +0 -0
run-7dukmcwd/checkpoint-400/tokenizer_config.json +55 -0
run-7dukmcwd/checkpoint-400/trainer_state.json +345 -0
run-7dukmcwd/checkpoint-400/training_args.bin +3 -0
run-7dukmcwd/checkpoint-400/vocab.txt +0 -0
run-7naa0m57/checkpoint-1200/config.json +26 -0
run-7naa0m57/checkpoint-1200/model.safetensors +3 -0
run-7naa0m57/checkpoint-1200/optimizer.pt +3 -0

README.md ADDED Viewed

	@@ -0,0 +1,70 @@

+---
+library_name: transformers
+license: apache-2.0
+base_model: distilbert-base-multilingual-cased
+tags:
+- generated_from_trainer
+metrics:
+- accuracy
+- recall
+- precision
+- f1
+model-index:
+- name: distilbert-base-multilingual-cased-hyper-matt
+  results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# distilbert-base-multilingual-cased-hyper-matt
+This model is a fine-tuned version of [distilbert-base-multilingual-cased](https://huggingface.co/distilbert-base-multilingual-cased) on the None dataset.
+It achieves the following results on the evaluation set:
+- Loss: 0.4924
+- Accuracy: 0.88
+- Recall: 0.7967
+- Precision: 0.8099
+- F1: 0.8033
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 1.0247874543132884e-05
+- train_batch_size: 4
+- eval_batch_size: 16
+- seed: 13
+- optimizer: Use adamw_torch with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
+- lr_scheduler_type: linear
+- num_epochs: 4
+### Training results
+| Training Loss | Epoch | Step | Validation Loss | Accuracy | Recall | Precision | F1     |
+|:-------------:|:-----:|:----:|:---------------:|:--------:|:------:|:---------:|:------:|
+| 0.1205        | 1.0   | 400  | 0.3566          | 0.875    | 0.7967 | 0.7967    | 0.7967 |
+| 0.1525        | 2.0   | 800  | 0.5043          | 0.855    | 0.8618 | 0.7211    | 0.7852 |
+| 0.341         | 3.0   | 1200 | 0.4811          | 0.865    | 0.8211 | 0.7594    | 0.7891 |
+| 0.1254        | 4.0   | 1600 | 0.4924          | 0.88     | 0.7967 | 0.8099    | 0.8033 |
+### Framework versions
+- Transformers 4.46.3
+- Pytorch 2.5.1+cu121
+- Datasets 3.2.0
+- Tokenizers 0.20.3

config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "_name_or_path": "distilbert-base-multilingual-cased",
+  "activation": "gelu",
+  "architectures": [
+    "DistilBertForSequenceClassification"
+  ],
+  "attention_dropout": 0.1,
+  "dim": 768,
+  "dropout": 0.1,
+  "hidden_dim": 3072,
+  "initializer_range": 0.02,
+  "max_position_embeddings": 512,
+  "model_type": "distilbert",
+  "n_heads": 12,
+  "n_layers": 6,
+  "output_past": true,
+  "pad_token_id": 0,
+  "problem_type": "single_label_classification",
+  "qa_dropout": 0.1,
+  "seq_classif_dropout": 0.2,
+  "sinusoidal_pos_embds": false,
+  "tie_weights_": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.3",
+  "vocab_size": 119547
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aeedea8476eaf900a429a1730c800bbe49d1da0ab3a93a30f49d11e1ddcb2525
+size 541317368

run-0i7sidmz/checkpoint-100/config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "_name_or_path": "distilbert-base-multilingual-cased",
+  "activation": "gelu",
+  "architectures": [
+    "DistilBertForSequenceClassification"
+  ],
+  "attention_dropout": 0.1,
+  "dim": 768,
+  "dropout": 0.1,
+  "hidden_dim": 3072,
+  "initializer_range": 0.02,
+  "max_position_embeddings": 512,
+  "model_type": "distilbert",
+  "n_heads": 12,
+  "n_layers": 6,
+  "output_past": true,
+  "pad_token_id": 0,
+  "problem_type": "single_label_classification",
+  "qa_dropout": 0.1,
+  "seq_classif_dropout": 0.2,
+  "sinusoidal_pos_embds": false,
+  "tie_weights_": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.3",
+  "vocab_size": 119547
+}

run-0i7sidmz/checkpoint-100/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:09cb87566740ddc29ee7cb0b2fe8ffeaba5d642f5be9217340698090fe24a859
+size 541317368

run-0i7sidmz/checkpoint-100/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4a669da35d5171f07d821ca4186b7946fc0c61708b935065f8727cad930506d3
+size 1082696890

run-0i7sidmz/checkpoint-100/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2b9e9384edf1155e453438261efb67244409e82607e8085254a001960694cb5e
+size 14244

run-0i7sidmz/checkpoint-100/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9c49099a4da9d1fc29a715d6d571d6600b786862df85ebb394237d7db97e4a38
+size 1064

run-0i7sidmz/checkpoint-100/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "cls_token": "[CLS]",
+  "mask_token": "[MASK]",
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "unk_token": "[UNK]"
+}

run-0i7sidmz/checkpoint-100/tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

run-0i7sidmz/checkpoint-100/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "[PAD]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "100": {
+      "content": "[UNK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "101": {
+      "content": "[CLS]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "102": {
+      "content": "[SEP]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "103": {
+      "content": "[MASK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "[CLS]",
+  "do_lower_case": false,
+  "mask_token": "[MASK]",
+  "model_max_length": 512,
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "strip_accents": null,
+  "tokenize_chinese_chars": true,
+  "tokenizer_class": "DistilBertTokenizer",
+  "unk_token": "[UNK]"
+}

run-0i7sidmz/checkpoint-100/trainer_state.json ADDED Viewed

	@@ -0,0 +1,123 @@

+{
+  "best_metric": 0.7607843137254902,
+  "best_model_checkpoint": "distilbert-base-multilingual-cased-hyper-matt/run-0i7sidmz/checkpoint-100",
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": true,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.1,
+      "grad_norm": 4.02569055557251,
+      "learning_rate": 5.021793617667588e-05,
+      "loss": 0.5846,
+      "step": 10
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 2.9225821495056152,
+      "learning_rate": 4.4638165490378564e-05,
+      "loss": 0.4403,
+      "step": 20
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 3.3444550037384033,
+      "learning_rate": 3.905839480408124e-05,
+      "loss": 0.6524,
+      "step": 30
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 3.390305280685425,
+      "learning_rate": 3.3478624117783916e-05,
+      "loss": 0.477,
+      "step": 40
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 4.890294551849365,
+      "learning_rate": 2.78988534314866e-05,
+      "loss": 0.4469,
+      "step": 50
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 3.794447183609009,
+      "learning_rate": 2.2319082745189282e-05,
+      "loss": 0.4012,
+      "step": 60
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 5.297450065612793,
+      "learning_rate": 1.6739312058891958e-05,
+      "loss": 0.396,
+      "step": 70
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 3.4959356784820557,
+      "learning_rate": 1.1159541372594641e-05,
+      "loss": 0.4017,
+      "step": 80
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 4.851554870605469,
+      "learning_rate": 5.5797706862973205e-06,
+      "loss": 0.4157,
+      "step": 90
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 4.022997856140137,
+      "learning_rate": 0.0,
+      "loss": 0.3782,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "eval_accuracy": 0.8475,
+      "eval_f1": 0.7607843137254902,
+      "eval_loss": 0.3398902416229248,
+      "eval_precision": 0.7348484848484849,
+      "eval_recall": 0.7886178861788617,
+      "eval_runtime": 1.5686,
+      "eval_samples_per_second": 255.004,
+      "eval_steps_per_second": 15.938,
+      "step": 100
+    }
+  ],
+  "logging_steps": 10,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 211815370450944.0,
+  "train_batch_size": 16,
+  "trial_name": null,
+  "trial_params": {
+    "_wandb": {},
+    "assignments": {},
+    "learning_rate": 5.57977068629732e-05,
+    "metric": "eval/loss",
+    "num_train_epochs": 1,
+    "per_device_train_batch_size": 16,
+    "seed": 28
+  }
+}

run-0i7sidmz/checkpoint-100/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c40a071b993997cf8dd6a013a689ae7e699b5cce64898ae6129d0b3b4cc55b62
+size 5240

run-0i7sidmz/checkpoint-100/vocab.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

run-0nyui8bu/checkpoint-25/config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "_name_or_path": "distilbert-base-multilingual-cased",
+  "activation": "gelu",
+  "architectures": [
+    "DistilBertForSequenceClassification"
+  ],
+  "attention_dropout": 0.1,
+  "dim": 768,
+  "dropout": 0.1,
+  "hidden_dim": 3072,
+  "initializer_range": 0.02,
+  "max_position_embeddings": 512,
+  "model_type": "distilbert",
+  "n_heads": 12,
+  "n_layers": 6,
+  "output_past": true,
+  "pad_token_id": 0,
+  "problem_type": "single_label_classification",
+  "qa_dropout": 0.1,
+  "seq_classif_dropout": 0.2,
+  "sinusoidal_pos_embds": false,
+  "tie_weights_": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.3",
+  "vocab_size": 119547
+}

run-0nyui8bu/checkpoint-25/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:378eccdb882964b83f6ba71406ff46a4e2ac412890df690c542bf96d7f9d5d99
+size 541317368

run-0nyui8bu/checkpoint-25/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c1f856965b59a002d06c61f92da7eb0a59ea3bf1e1c4986050fb9fbc59523192
+size 1082696890

run-0nyui8bu/checkpoint-25/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d9b65e6ac01ec50489b575aacd029c5ff17ebd10c7296199f8ece046469b8efb
+size 14308

run-0nyui8bu/checkpoint-25/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:afb8ff5dcf306be3f665d22d4db51cd102205d8c962baefb7289d649d2dce275
+size 1064

run-0nyui8bu/checkpoint-25/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "cls_token": "[CLS]",
+  "mask_token": "[MASK]",
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "unk_token": "[UNK]"
+}

run-0nyui8bu/checkpoint-25/tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

run-0nyui8bu/checkpoint-25/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "[PAD]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "100": {
+      "content": "[UNK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "101": {
+      "content": "[CLS]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "102": {
+      "content": "[SEP]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "103": {
+      "content": "[MASK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "[CLS]",
+  "do_lower_case": false,
+  "mask_token": "[MASK]",
+  "model_max_length": 512,
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "strip_accents": null,
+  "tokenize_chinese_chars": true,
+  "tokenizer_class": "DistilBertTokenizer",
+  "unk_token": "[UNK]"
+}

run-0nyui8bu/checkpoint-25/trainer_state.json ADDED Viewed

	@@ -0,0 +1,67 @@

+{
+  "best_metric": 0.7364016736401674,
+  "best_model_checkpoint": "distilbert-base-multilingual-cased-hyper-matt/run-0nyui8bu/checkpoint-25",
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 25,
+  "is_hyper_param_search": true,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.4,
+      "grad_norm": 1.2732082605361938,
+      "learning_rate": 5.6232629293362164e-05,
+      "loss": 0.5343,
+      "step": 10
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 3.3999197483062744,
+      "learning_rate": 1.8744209764454056e-05,
+      "loss": 0.4344,
+      "step": 20
+    },
+    {
+      "epoch": 1.0,
+      "eval_accuracy": 0.8425,
+      "eval_f1": 0.7364016736401674,
+      "eval_loss": 0.3744986355304718,
+      "eval_precision": 0.7586206896551724,
+      "eval_recall": 0.7154471544715447,
+      "eval_runtime": 1.5686,
+      "eval_samples_per_second": 255.002,
+      "eval_steps_per_second": 15.938,
+      "step": 25
+    }
+  ],
+  "logging_steps": 10,
+  "max_steps": 25,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 169558270279680.0,
+  "train_batch_size": 64,
+  "trial_name": null,
+  "trial_params": {
+    "_wandb": {},
+    "assignments": {},
+    "learning_rate": 9.372104882227028e-05,
+    "metric": "eval/loss",
+    "num_train_epochs": 1,
+    "per_device_train_batch_size": 64,
+    "seed": 37
+  }
+}

run-0nyui8bu/checkpoint-25/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bce4c692072b031db6e9fc87725db24c8a929ad389230d3c414d98cc04b83cf7
+size 5240

run-0nyui8bu/checkpoint-25/vocab.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

run-4r5wbvc3/checkpoint-50/config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "_name_or_path": "distilbert-base-multilingual-cased",
+  "activation": "gelu",
+  "architectures": [
+    "DistilBertForSequenceClassification"
+  ],
+  "attention_dropout": 0.1,
+  "dim": 768,
+  "dropout": 0.1,
+  "hidden_dim": 3072,
+  "initializer_range": 0.02,
+  "max_position_embeddings": 512,
+  "model_type": "distilbert",
+  "n_heads": 12,
+  "n_layers": 6,
+  "output_past": true,
+  "pad_token_id": 0,
+  "problem_type": "single_label_classification",
+  "qa_dropout": 0.1,
+  "seq_classif_dropout": 0.2,
+  "sinusoidal_pos_embds": false,
+  "tie_weights_": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.3",
+  "vocab_size": 119547
+}

run-4r5wbvc3/checkpoint-50/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0f82f8fac828a11d8fca5af71307d1126d654ee816f86442f92592da36ca5c26
+size 541317368

run-4r5wbvc3/checkpoint-50/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2e1274b02a1ae6f3339c0eb8e7cf85ad9e9a67f152174449e88df286f6a87afb
+size 1082696890

run-4r5wbvc3/checkpoint-50/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a7919d1f388faa1a24e8c5f592a11d4313bb7a759b42c854cd52b920ab4f9955
+size 14244

run-4r5wbvc3/checkpoint-50/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e00b842047d002138977aef49691fc0fb19d9b46b1cb805864154c324450686c
+size 1064

run-4r5wbvc3/checkpoint-50/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "cls_token": "[CLS]",
+  "mask_token": "[MASK]",
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "unk_token": "[UNK]"
+}

run-4r5wbvc3/checkpoint-50/tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

run-4r5wbvc3/checkpoint-50/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "[PAD]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "100": {
+      "content": "[UNK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "101": {
+      "content": "[CLS]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "102": {
+      "content": "[SEP]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "103": {
+      "content": "[MASK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "[CLS]",
+  "do_lower_case": false,
+  "mask_token": "[MASK]",
+  "model_max_length": 512,
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "strip_accents": null,
+  "tokenize_chinese_chars": true,
+  "tokenizer_class": "DistilBertTokenizer",
+  "unk_token": "[UNK]"
+}

run-4r5wbvc3/checkpoint-50/trainer_state.json ADDED Viewed

	@@ -0,0 +1,88 @@

+{
+  "best_metric": 0.582995951417004,
+  "best_model_checkpoint": "distilbert-base-multilingual-cased-hyper-matt/run-4r5wbvc3/checkpoint-50",
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 50,
+  "is_hyper_param_search": true,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.2,
+      "grad_norm": 1.2984086275100708,
+      "learning_rate": 2.8254267437810627e-05,
+      "loss": 0.6014,
+      "step": 10
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 1.4973173141479492,
+      "learning_rate": 2.119070057835797e-05,
+      "loss": 0.5197,
+      "step": 20
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 1.660354733467102,
+      "learning_rate": 1.4127133718905313e-05,
+      "loss": 0.4239,
+      "step": 30
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 1.9838685989379883,
+      "learning_rate": 7.063566859452657e-06,
+      "loss": 0.4782,
+      "step": 40
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 2.0562429428100586,
+      "learning_rate": 0.0,
+      "loss": 0.4456,
+      "step": 50
+    },
+    {
+      "epoch": 1.0,
+      "eval_accuracy": 0.7425,
+      "eval_f1": 0.582995951417004,
+      "eval_loss": 0.4402432143688202,
+      "eval_precision": 0.5806451612903226,
+      "eval_recall": 0.5853658536585366,
+      "eval_runtime": 1.5859,
+      "eval_samples_per_second": 252.225,
+      "eval_steps_per_second": 15.764,
+      "step": 50
+    }
+  ],
+  "logging_steps": 10,
+  "max_steps": 50,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 211815370450944.0,
+  "train_batch_size": 32,
+  "trial_name": null,
+  "trial_params": {
+    "_wandb": {},
+    "assignments": {},
+    "learning_rate": 3.531783429726328e-05,
+    "metric": "eval/loss",
+    "num_train_epochs": 1,
+    "per_device_train_batch_size": 32,
+    "seed": 5
+  }
+}

run-4r5wbvc3/checkpoint-50/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4f629cd4e45120ede64c8be90f1d697ea1dce5976fb5d20883afc523455bec9d
+size 5240

run-4r5wbvc3/checkpoint-50/vocab.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

run-7dukmcwd/checkpoint-400/config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "_name_or_path": "distilbert-base-multilingual-cased",
+  "activation": "gelu",
+  "architectures": [
+    "DistilBertForSequenceClassification"
+  ],
+  "attention_dropout": 0.1,
+  "dim": 768,
+  "dropout": 0.1,
+  "hidden_dim": 3072,
+  "initializer_range": 0.02,
+  "max_position_embeddings": 512,
+  "model_type": "distilbert",
+  "n_heads": 12,
+  "n_layers": 6,
+  "output_past": true,
+  "pad_token_id": 0,
+  "problem_type": "single_label_classification",
+  "qa_dropout": 0.1,
+  "seq_classif_dropout": 0.2,
+  "sinusoidal_pos_embds": false,
+  "tie_weights_": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.3",
+  "vocab_size": 119547
+}

run-7dukmcwd/checkpoint-400/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6f00836fe55e04178078dc75a663d7ac8df310c9c4985b909d055a2ab25977b1
+size 541317368

run-7dukmcwd/checkpoint-400/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0068aafd305005d87db7a848f38cc6383002f8b459697deccdb000ea429cd6f9
+size 1082696890

run-7dukmcwd/checkpoint-400/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:19cd3ce56ddc2de67afd95c20035824951a67afe7f7edaa24182b8adbd2860b3
+size 14244

run-7dukmcwd/checkpoint-400/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4bf54c78186771ba688aae61205c12d87f3cc71868624e420132584e5c670163
+size 1064

run-7dukmcwd/checkpoint-400/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "cls_token": "[CLS]",
+  "mask_token": "[MASK]",
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "unk_token": "[UNK]"
+}

run-7dukmcwd/checkpoint-400/tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

run-7dukmcwd/checkpoint-400/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "[PAD]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "100": {
+      "content": "[UNK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "101": {
+      "content": "[CLS]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "102": {
+      "content": "[SEP]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "103": {
+      "content": "[MASK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "cls_token": "[CLS]",
+  "do_lower_case": false,
+  "mask_token": "[MASK]",
+  "model_max_length": 512,
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "strip_accents": null,
+  "tokenize_chinese_chars": true,
+  "tokenizer_class": "DistilBertTokenizer",
+  "unk_token": "[UNK]"
+}

run-7dukmcwd/checkpoint-400/trainer_state.json ADDED Viewed

	@@ -0,0 +1,345 @@

+{
+  "best_metric": 0.8032786885245902,
+  "best_model_checkpoint": "distilbert-base-multilingual-cased-hyper-matt/run-7dukmcwd/checkpoint-400",
+  "epoch": 2.0,
+  "eval_steps": 500,
+  "global_step": 400,
+  "is_hyper_param_search": true,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.05,
+      "grad_norm": 2.6303067207336426,
+      "learning_rate": 3.5703870009677385e-05,
+      "loss": 0.6604,
+      "step": 10
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 2.9712932109832764,
+      "learning_rate": 3.47883861632754e-05,
+      "loss": 0.5592,
+      "step": 20
+    },
+    {
+      "epoch": 0.15,
+      "grad_norm": 3.1559805870056152,
+      "learning_rate": 3.387290231687342e-05,
+      "loss": 0.4817,
+      "step": 30
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 8.0116548538208,
+      "learning_rate": 3.295741847047143e-05,
+      "loss": 0.4318,
+      "step": 40
+    },
+    {
+      "epoch": 0.25,
+      "grad_norm": 19.13960075378418,
+      "learning_rate": 3.204193462406945e-05,
+      "loss": 0.5086,
+      "step": 50
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 13.061105728149414,
+      "learning_rate": 3.1126450777667465e-05,
+      "loss": 0.527,
+      "step": 60
+    },
+    {
+      "epoch": 0.35,
+      "grad_norm": 12.58017635345459,
+      "learning_rate": 3.0210966931265478e-05,
+      "loss": 0.4828,
+      "step": 70
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 5.202507495880127,
+      "learning_rate": 2.9295483084863497e-05,
+      "loss": 0.4613,
+      "step": 80
+    },
+    {
+      "epoch": 0.45,
+      "grad_norm": 5.813719749450684,
+      "learning_rate": 2.8379999238461513e-05,
+      "loss": 0.4274,
+      "step": 90
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 8.482573509216309,
+      "learning_rate": 2.746451539205953e-05,
+      "loss": 0.4159,
+      "step": 100
+    },
+    {
+      "epoch": 0.55,
+      "grad_norm": 9.623395919799805,
+      "learning_rate": 2.654903154565754e-05,
+      "loss": 0.3306,
+      "step": 110
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 5.012599468231201,
+      "learning_rate": 2.5633547699255557e-05,
+      "loss": 0.403,
+      "step": 120
+    },
+    {
+      "epoch": 0.65,
+      "grad_norm": 3.260857343673706,
+      "learning_rate": 2.4718063852853577e-05,
+      "loss": 0.3636,
+      "step": 130
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 18.5455379486084,
+      "learning_rate": 2.3802580006451593e-05,
+      "loss": 0.3621,
+      "step": 140
+    },
+    {
+      "epoch": 0.75,
+      "grad_norm": 3.035172700881958,
+      "learning_rate": 2.2887096160049605e-05,
+      "loss": 0.376,
+      "step": 150
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 6.068894386291504,
+      "learning_rate": 2.197161231364762e-05,
+      "loss": 0.3452,
+      "step": 160
+    },
+    {
+      "epoch": 0.85,
+      "grad_norm": 3.0021464824676514,
+      "learning_rate": 2.1056128467245637e-05,
+      "loss": 0.3577,
+      "step": 170
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 3.3914709091186523,
+      "learning_rate": 2.0140644620843656e-05,
+      "loss": 0.4271,
+      "step": 180
+    },
+    {
+      "epoch": 0.95,
+      "grad_norm": 8.317371368408203,
+      "learning_rate": 1.922516077444167e-05,
+      "loss": 0.289,
+      "step": 190
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 0.967580258846283,
+      "learning_rate": 1.8309676928039685e-05,
+      "loss": 0.3738,
+      "step": 200
+    },
+    {
+      "epoch": 1.0,
+      "eval_accuracy": 0.87,
+      "eval_f1": 0.7657657657657657,
+      "eval_loss": 0.32556435465812683,
+      "eval_precision": 0.8585858585858586,
+      "eval_recall": 0.6910569105691057,
+      "eval_runtime": 1.5112,
+      "eval_samples_per_second": 264.697,
+      "eval_steps_per_second": 16.544,
+      "step": 200
+    },
+    {
+      "epoch": 1.05,
+      "grad_norm": 4.126100540161133,
+      "learning_rate": 1.73941930816377e-05,
+      "loss": 0.2486,
+      "step": 210
+    },
+    {
+      "epoch": 1.1,
+      "grad_norm": 5.108118057250977,
+      "learning_rate": 1.6478709235235717e-05,
+      "loss": 0.3034,
+      "step": 220
+    },
+    {
+      "epoch": 1.15,
+      "grad_norm": 11.375035285949707,
+      "learning_rate": 1.5563225388833733e-05,
+      "loss": 0.1486,
+      "step": 230
+    },
+    {
+      "epoch": 1.2,
+      "grad_norm": 8.199675559997559,
+      "learning_rate": 1.4647741542431748e-05,
+      "loss": 0.327,
+      "step": 240
+    },
+    {
+      "epoch": 1.25,
+      "grad_norm": 4.900712013244629,
+      "learning_rate": 1.3732257696029764e-05,
+      "loss": 0.2753,
+      "step": 250
+    },
+    {
+      "epoch": 1.3,
+      "grad_norm": 0.31448882818222046,
+      "learning_rate": 1.2816773849627779e-05,
+      "loss": 0.2028,
+      "step": 260
+    },
+    {
+      "epoch": 1.35,
+      "grad_norm": 0.3391319513320923,
+      "learning_rate": 1.1901290003225796e-05,
+      "loss": 0.2992,
+      "step": 270
+    },
+    {
+      "epoch": 1.4,
+      "grad_norm": 20.60084342956543,
+      "learning_rate": 1.098580615682381e-05,
+      "loss": 0.4703,
+      "step": 280
+    },
+    {
+      "epoch": 1.45,
+      "grad_norm": 7.974413871765137,
+      "learning_rate": 1.0070322310421828e-05,
+      "loss": 0.2649,
+      "step": 290
+    },
+    {
+      "epoch": 1.5,
+      "grad_norm": 11.488137245178223,
+      "learning_rate": 9.154838464019842e-06,
+      "loss": 0.438,
+      "step": 300
+    },
+    {
+      "epoch": 1.55,
+      "grad_norm": 0.5850751399993896,
+      "learning_rate": 8.239354617617858e-06,
+      "loss": 0.1704,
+      "step": 310
+    },
+    {
+      "epoch": 1.6,
+      "grad_norm": 3.258329391479492,
+      "learning_rate": 7.323870771215874e-06,
+      "loss": 0.226,
+      "step": 320
+    },
+    {
+      "epoch": 1.65,
+      "grad_norm": 6.117366790771484,
+      "learning_rate": 6.408386924813889e-06,
+      "loss": 0.2219,
+      "step": 330
+    },
+    {
+      "epoch": 1.7,
+      "grad_norm": 28.112499237060547,
+      "learning_rate": 5.492903078411905e-06,
+      "loss": 0.2595,
+      "step": 340
+    },
+    {
+      "epoch": 1.75,
+      "grad_norm": 15.969998359680176,
+      "learning_rate": 4.577419232009921e-06,
+      "loss": 0.3709,
+      "step": 350
+    },
+    {
+      "epoch": 1.8,
+      "grad_norm": 0.6372332572937012,
+      "learning_rate": 3.661935385607937e-06,
+      "loss": 0.2678,
+      "step": 360
+    },
+    {
+      "epoch": 1.85,
+      "grad_norm": 0.20131894946098328,
+      "learning_rate": 2.7464515392059526e-06,
+      "loss": 0.4027,
+      "step": 370
+    },
+    {
+      "epoch": 1.9,
+      "grad_norm": 12.212553024291992,
+      "learning_rate": 1.8309676928039686e-06,
+      "loss": 0.3156,
+      "step": 380
+    },
+    {
+      "epoch": 1.95,
+      "grad_norm": 0.19253325462341309,
+      "learning_rate": 9.154838464019843e-07,
+      "loss": 0.1869,
+      "step": 390
+    },
+    {
+      "epoch": 2.0,
+      "grad_norm": 14.685490608215332,
+      "learning_rate": 0.0,
+      "loss": 0.1768,
+      "step": 400
+    },
+    {
+      "epoch": 2.0,
+      "eval_accuracy": 0.88,
+      "eval_f1": 0.8032786885245902,
+      "eval_loss": 0.3457476794719696,
+      "eval_precision": 0.8099173553719008,
+      "eval_recall": 0.7967479674796748,
+      "eval_runtime": 1.5723,
+      "eval_samples_per_second": 254.401,
+      "eval_steps_per_second": 15.9,
+      "step": 400
+    }
+  ],
+  "logging_steps": 10,
+  "max_steps": 400,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 2,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 423630740901888.0,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": {
+    "_wandb": {},
+    "assignments": {},
+    "learning_rate": 3.661935385607937e-05,
+    "metric": "eval/loss",
+    "num_train_epochs": 2,
+    "per_device_train_batch_size": 8,
+    "seed": 36
+  }
+}

run-7dukmcwd/checkpoint-400/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:11e58247026a6e43bf235eb24f72fa34866b220551a67333de2b1357c79bea2b
+size 5240

run-7dukmcwd/checkpoint-400/vocab.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

run-7naa0m57/checkpoint-1200/config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "_name_or_path": "distilbert-base-multilingual-cased",
+  "activation": "gelu",
+  "architectures": [
+    "DistilBertForSequenceClassification"
+  ],
+  "attention_dropout": 0.1,
+  "dim": 768,
+  "dropout": 0.1,
+  "hidden_dim": 3072,
+  "initializer_range": 0.02,
+  "max_position_embeddings": 512,
+  "model_type": "distilbert",
+  "n_heads": 12,
+  "n_layers": 6,
+  "output_past": true,
+  "pad_token_id": 0,
+  "problem_type": "single_label_classification",
+  "qa_dropout": 0.1,
+  "seq_classif_dropout": 0.2,
+  "sinusoidal_pos_embds": false,
+  "tie_weights_": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.46.3",
+  "vocab_size": 119547
+}

run-7naa0m57/checkpoint-1200/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:28bf6271d2a0afd7f0c29b1d663fb113babdf2590195af26f6575ff76315aed8
+size 541317368

run-7naa0m57/checkpoint-1200/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4b5236aff9ca64270c3ca8cf6183ee51a8a65ea87452421b45b79466142210a6
+size 1082696890