Training in progress, step 1000

Files changed (14) hide show

.gitignore ADDED Viewed

	@@ -0,0 +1 @@


1	+ checkpoint-*/

config.json ADDED Viewed

+{
+  "_name_or_path": "sberbank-ai/rugpt3medium_based_on_gpt2",
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.1,
+  "bos_token_id": 50256,
+  "embd_pdrop": 0.1,
+  "eos_token_id": 50256,
+  "id2label": {
+    "0": "LABEL_0"
+  },
+  "initializer_range": 0.02,
+  "label2id": {
+    "LABEL_0": 0
+  },
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_ctx": 2048,
+  "n_embd": 1024,
+  "n_head": 16,
+  "n_inner": null,
+  "n_layer": 24,
+  "n_positions": 2048,
+  "n_special": 0,
+  "output_past": true,
+  "predict_special_tokens": true,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.1,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.27.3",
+  "use_cache": true,
+  "vocab_size": 50257
+}

last-checkpoint/config.json ADDED Viewed

+{
+  "_name_or_path": "sberbank-ai/rugpt3medium_based_on_gpt2",
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.1,
+  "bos_token_id": 50256,
+  "embd_pdrop": 0.1,
+  "eos_token_id": 50256,
+  "id2label": {
+    "0": "LABEL_0"
+  },
+  "initializer_range": 0.02,
+  "label2id": {
+    "LABEL_0": 0
+  },
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_ctx": 2048,
+  "n_embd": 1024,
+  "n_head": 16,
+  "n_inner": null,
+  "n_layer": 24,
+  "n_positions": 2048,
+  "n_special": 0,
+  "output_past": true,
+  "predict_special_tokens": true,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.1,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.27.3",
+  "use_cache": true,
+  "vocab_size": 50257
+}

last-checkpoint/generation_config.json ADDED Viewed

+{
+  "_from_model_config": true,
+  "bos_token_id": 50256,
+  "eos_token_id": 50256,
+  "transformers_version": "4.27.3"
+}

last-checkpoint/optimizer.pt ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:5845db90c819ba4a6c7efeb0b2dcbbebd50c79ba1d34e3e084abcb212d94828a
+size 2847145157

last-checkpoint/pytorch_model.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:ca54c56bdbf19362dc38643bb215b49d632e199bcbed91ab6caf1417f01a57d7
+size 1524261149

last-checkpoint/rng_state.pth ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e76f5afdcb7fcb4bfc4847536c3e3a04a3d753394ce857c4f72002f0b303acf
+size 14575

last-checkpoint/scheduler.pt ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:2dd5affc042bd27e2a30b219358a20190c5a1ffcb64f301899cb976a09f547ad
+size 627

last-checkpoint/trainer_state.json ADDED Viewed

+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.5002501250625313,
+  "global_step": 1000,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.25,
+      "learning_rate": 1.8999499749874938e-05,
+      "loss": 3.2627,
+      "step": 500
+    },
+    {
+      "epoch": 0.5,
+      "learning_rate": 1.7998999499749875e-05,
+      "loss": 3.2043,
+      "step": 1000
+    },
+    {
+      "epoch": 0.5,
+      "eval_loss": 3.152320146560669,
+      "eval_runtime": 135.0377,
+      "eval_samples_per_second": 15.677,
+      "eval_steps_per_second": 2.614,
+      "step": 1000
+    }
+  ],
+  "max_steps": 9995,
+  "num_train_epochs": 5,
+  "total_flos": 2829634928640000.0,
+  "trial_name": null,
+  "trial_params": null
+}

last-checkpoint/training_args.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:175fa8453377bf4a0c9a3ff9526282460c0fde24b49844814ef9cd85c3698074
+size 3515

pytorch_model.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:ca54c56bdbf19362dc38643bb215b49d632e199bcbed91ab6caf1417f01a57d7
+size 1524261149

runs/Mar24_13-56-57_9891ffea5fdc/1679666221.1599746/events.out.tfevents.1679666221.9891ffea5fdc.675.1 ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:b189a623277eecc7515a5d4a5edd93e0ff40b35135f1528df71fc112e44728d5
+size 5754

runs/Mar24_13-56-57_9891ffea5fdc/events.out.tfevents.1679666221.9891ffea5fdc.675.0 ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:b9663ca43f9fe5b2bc50d936df2482f84b127c637922a236ca1770a9703a38f8
+size 4740

training_args.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:175fa8453377bf4a0c9a3ff9526282460c0fde24b49844814ef9cd85c3698074
+size 3515