0xsuid commited on Feb 1, 2023

Commit

5ea9a88

•

1 Parent(s): 8055c91

Model save

Browse files

Files changed (21) hide show

last-checkpoint/generation_config.json → generation_config.json +0 -0
last-checkpoint/config.json +0 -74
last-checkpoint/global_step1143/mp_rank_00_model_states.pt +0 -3
last-checkpoint/global_step1143/zero_pp_rank_0_mp_rank_00_optim_states.pt +0 -3
last-checkpoint/global_step1143/zero_pp_rank_1_mp_rank_00_optim_states.pt +0 -3
last-checkpoint/global_step1143/zero_pp_rank_2_mp_rank_00_optim_states.pt +0 -3
last-checkpoint/global_step1143/zero_pp_rank_3_mp_rank_00_optim_states.pt +0 -3
last-checkpoint/latest +0 -1
last-checkpoint/merges.txt +0 -0
last-checkpoint/pytorch_model.bin +0 -3
last-checkpoint/rng_state_0.pth +0 -3
last-checkpoint/rng_state_1.pth +0 -3
last-checkpoint/rng_state_2.pth +0 -3
last-checkpoint/rng_state_3.pth +0 -3
last-checkpoint/special_tokens_map.json +0 -6
last-checkpoint/tokenizer.json +0 -0
last-checkpoint/tokenizer_config.json +0 -34
last-checkpoint/trainer_state.json +0 -1390
last-checkpoint/training_args.bin +0 -3
last-checkpoint/vocab.json +0 -0
last-checkpoint/zero_to_fp32.py +0 -482

last-checkpoint/generation_config.json → generation_config.json RENAMED Viewed

File without changes

last-checkpoint/config.json DELETED Viewed

@@ -1,74 +0,0 @@
-{
-  "_name_or_path": "EleutherAI/gpt-neo-1.3B",
-  "activation_function": "gelu_new",
-  "architectures": [
-    "GPTNeoForCausalLM"
-  ],
-  "attention_dropout": 0,
-  "attention_layers": [
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local",
-    "global",
-    "local"
-  ],
-  "attention_types": [
-    [
-      [
-        "global",
-        "local"
-      ],
-      12
-    ]
-  ],
-  "bos_token_id": 50256,
-  "embed_dropout": 0,
-  "eos_token_id": 50256,
-  "gradient_checkpointing": false,
-  "hidden_size": 2048,
-  "initializer_range": 0.02,
-  "intermediate_size": null,
-  "layer_norm_epsilon": 1e-05,
-  "max_position_embeddings": 2048,
-  "model_type": "gpt_neo",
-  "num_heads": 16,
-  "num_layers": 24,
-  "resid_dropout": 0,
-  "summary_activation": null,
-  "summary_first_dropout": 0.1,
-  "summary_proj_to_labels": true,
-  "summary_type": "cls_index",
-  "summary_use_proj": true,
-  "task_specific_params": {
-    "text-generation": {
-      "do_sample": true,
-      "max_length": 50,
-      "temperature": 0.9
-    }
-  },
-  "tokenizer_class": "GPT2Tokenizer",
-  "torch_dtype": "float32",
-  "transformers_version": "4.26.0",
-  "use_cache": false,
-  "vocab_size": 50257,
-  "window_size": 256
-}

last-checkpoint/global_step1143/mp_rank_00_model_states.pt DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:b30eb2fb488eaec8a27bb0b26a171963d3112a31c76479ce8dfdd718e9567bf2
-size 5363072554

last-checkpoint/global_step1143/zero_pp_rank_0_mp_rank_00_optim_states.pt DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:d3e6445cc536442d208833ad59863f984bc71140490a8cd3e18575006fcdd4e1
-size 3946735038

last-checkpoint/global_step1143/zero_pp_rank_1_mp_rank_00_optim_states.pt DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:11f87ca6a8d81d91ce9669ce0a0b132e16e31c589165b9dfc0f95bec607a1ce3
-size 3946736318

last-checkpoint/global_step1143/zero_pp_rank_2_mp_rank_00_optim_states.pt DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:076063a2d14362bcc0562f87c7d5497798acb4d6078c32ba220db1ed473f33c8
-size 3946737086

last-checkpoint/global_step1143/zero_pp_rank_3_mp_rank_00_optim_states.pt DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:e5e8189f9e1fe3d7e7bb628368a866880f0176b1f62f4c271d4ccf25c9e8e98a
-size 3946736574

last-checkpoint/latest DELETED Viewed

	@@ -1 +0,0 @@
1	- global_step1143

last-checkpoint/merges.txt DELETED Viewed

The diff for this file is too large to render. See raw diff

last-checkpoint/pytorch_model.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:d6d85bffffb2fe97ca10f0460ebc0b029a11e6d606ade9c54de58bcb6de72ec8
-size 5363024236

last-checkpoint/rng_state_0.pth DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:2f6bc3b332b1d7b34dd8e7d7ed0389c868155059ddb1d908e9ac3feb6672b23c
-size 14583

last-checkpoint/rng_state_1.pth DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:8de5e0c7dadcd828a8d62fffc136e170202022509240a895985c7bc45cabbced
-size 14583

last-checkpoint/rng_state_2.pth DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:3c420d12d8aa09a561480241f19154d4aedd8a866de54ed145d69f860bae6f94
-size 14583

last-checkpoint/rng_state_3.pth DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:ec873fd7c31f869e7956f098c0d1e17d2296924b3c55e4971a059dd097690b6f
-size 14583

last-checkpoint/special_tokens_map.json DELETED Viewed

@@ -1,6 +0,0 @@
-{
-  "bos_token": "<|endoftext|>",
-  "eos_token": "<|endoftext|>",
-  "pad_token": "<|endoftext|>",
-  "unk_token": "<|endoftext|>"
-}

last-checkpoint/tokenizer.json DELETED Viewed

The diff for this file is too large to render. See raw diff

last-checkpoint/tokenizer_config.json DELETED Viewed

@@ -1,34 +0,0 @@
-{
-  "add_bos_token": false,
-  "add_prefix_space": false,
-  "bos_token": {
-    "__type": "AddedToken",
-    "content": "<|endoftext|>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  },
-  "eos_token": {
-    "__type": "AddedToken",
-    "content": "<|endoftext|>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  },
-  "errors": "replace",
-  "model_max_length": 2048,
-  "name_or_path": "EleutherAI/gpt-neo-1.3B",
-  "pad_token": null,
-  "special_tokens_map_file": null,
-  "tokenizer_class": "GPT2Tokenizer",
-  "unk_token": {
-    "__type": "AddedToken",
-    "content": "<|endoftext|>",
-    "lstrip": false,
-    "normalized": true,
-    "rstrip": false,
-    "single_word": false
-  }
-}

last-checkpoint/trainer_state.json DELETED Viewed

@@ -1,1390 +0,0 @@
-{
-  "best_metric": null,
-  "best_model_checkpoint": null,
-  "epoch": 4.996176952484981,
-  "global_step": 1140,
-  "is_hyper_param_search": false,
-  "is_local_process_zero": true,
-  "is_world_process_zero": true,
-  "log_history": [
-    {
-      "epoch": 0.0,
-      "learning_rate": 0.0,
-      "loss": 5.4178,
-      "step": 1
-    },
-    {
-      "epoch": 0.02,
-      "learning_rate": 1.294882868674145e-05,
-      "loss": 0.446,
-      "step": 5
-    },
-    {
-      "epoch": 0.04,
-      "learning_rate": 1.852558565662928e-05,
-      "loss": 0.3922,
-      "step": 10
-    },
-    {
-      "epoch": 0.07,
-      "learning_rate": 2.1787779359648994e-05,
-      "loss": 0.3732,
-      "step": 15
-    },
-    {
-      "epoch": 0.09,
-      "learning_rate": 2.41023426265171e-05,
-      "loss": 0.3591,
-      "step": 20
-    },
-    {
-      "epoch": 0.11,
-      "learning_rate": 2.58976573734829e-05,
-      "loss": 0.3384,
-      "step": 25
-    },
-    {
-      "epoch": 0.13,
-      "learning_rate": 2.7364536329536817e-05,
-      "loss": 0.3291,
-      "step": 30
-    },
-    {
-      "epoch": 0.15,
-      "learning_rate": 2.8604764815275082e-05,
-      "loss": 0.326,
-      "step": 35
-    },
-    {
-      "epoch": 0.17,
-      "learning_rate": 2.9679099596404923e-05,
-      "loss": 0.3234,
-      "step": 40
-    },
-    {
-      "epoch": 0.2,
-      "learning_rate": 3.0626730032556536e-05,
-      "loss": 0.3089,
-      "step": 45
-    },
-    {
-      "epoch": 0.22,
-      "learning_rate": 3.147441434337073e-05,
-      "loss": 0.2995,
-      "step": 50
-    },
-    {
-      "epoch": 0.24,
-      "learning_rate": 3.224123807782732e-05,
-      "loss": 0.2933,
-      "step": 55
-    },
-    {
-      "epoch": 0.26,
-      "learning_rate": 3.294129329942464e-05,
-      "loss": 0.2777,
-      "step": 60
-    },
-    {
-      "epoch": 0.28,
-      "learning_rate": 3.358528167653452e-05,
-      "loss": 0.2757,
-      "step": 65
-    },
-    {
-      "epoch": 0.31,
-      "learning_rate": 3.4181521785162905e-05,
-      "loss": 0.2674,
-      "step": 70
-    },
-    {
-      "epoch": 0.33,
-      "learning_rate": 3.473660804639045e-05,
-      "loss": 0.258,
-      "step": 75
-    },
-    {
-      "epoch": 0.35,
-      "learning_rate": 3.525585656629274e-05,
-      "loss": 0.2524,
-      "step": 80
-    },
-    {
-      "epoch": 0.37,
-      "learning_rate": 3.574361557584177e-05,
-      "loss": 0.2498,
-      "step": 85
-    },
-    {
-      "epoch": 0.39,
-      "learning_rate": 3.620348700244436e-05,
-      "loss": 0.2412,
-      "step": 90
-    },
-    {
-      "epoch": 0.42,
-      "learning_rate": 3.6638488054916214e-05,
-      "loss": 0.231,
-      "step": 95
-    },
-    {
-      "epoch": 0.44,
-      "learning_rate": 3.705117131325856e-05,
-      "loss": 0.227,
-      "step": 100
-    },
-    {
-      "epoch": 0.46,
-      "learning_rate": 3.7443715488182624e-05,
-      "loss": 0.219,
-      "step": 105
-    },
-    {
-      "epoch": 0.48,
-      "learning_rate": 3.781799504771514e-05,
-      "loss": 0.2189,
-      "step": 110
-    },
-    {
-      "epoch": 0.5,
-      "learning_rate": 3.81756343539018e-05,
-      "loss": 0.2148,
-      "step": 115
-    },
-    {
-      "epoch": 0.52,
-      "learning_rate": 3.851805026931246e-05,
-      "loss": 0.2054,
-      "step": 120
-    },
-    {
-      "epoch": 0.55,
-      "learning_rate": 3.8846486060224364e-05,
-      "loss": 0.2012,
-      "step": 125
-    },
-    {
-      "epoch": 0.57,
-      "learning_rate": 3.916203864642234e-05,
-      "loss": 0.1942,
-      "step": 130
-    },
-    {
-      "epoch": 0.59,
-      "learning_rate": 3.946568070546408e-05,
-      "loss": 0.1882,
-      "step": 135
-    },
-    {
-      "epoch": 0.61,
-      "learning_rate": 3.975827875505073e-05,
-      "loss": 0.1829,
-      "step": 140
-    },
-    {
-      "epoch": 0.63,
-      "learning_rate": 4.004060806090172e-05,
-      "loss": 0.1845,
-      "step": 145
-    },
-    {
-      "epoch": 0.66,
-      "learning_rate": 4.031336501627827e-05,
-      "loss": 0.1678,
-      "step": 150
-    },
-    {
-      "epoch": 0.68,
-      "learning_rate": 4.0577177490884e-05,
-      "loss": 0.1798,
-      "step": 155
-    },
-    {
-      "epoch": 0.7,
-      "learning_rate": 4.0832613536180565e-05,
-      "loss": 0.1735,
-      "step": 160
-    },
-    {
-      "epoch": 0.72,
-      "learning_rate": 4.1080188750734856e-05,
-      "loss": 0.176,
-      "step": 165
-    },
-    {
-      "epoch": 0.74,
-      "learning_rate": 4.1320372545729594e-05,
-      "loss": 0.1662,
-      "step": 170
-    },
-    {
-      "epoch": 0.76,
-      "learning_rate": 4.155359350201654e-05,
-      "loss": 0.1565,
-      "step": 175
-    },
-    {
-      "epoch": 0.79,
-      "learning_rate": 4.178024397233218e-05,
-      "loss": 0.1574,
-      "step": 180
-    },
-    {
-      "epoch": 0.81,
-      "learning_rate": 4.200068405281827e-05,
-      "loss": 0.1549,
-      "step": 185
-    },
-    {
-      "epoch": 0.83,
-      "learning_rate": 4.221524502480404e-05,
-      "loss": 0.1573,
-      "step": 190
-    },
-    {
-      "epoch": 0.85,
-      "learning_rate": 4.242423234944206e-05,
-      "loss": 0.1477,
-      "step": 195
-    },
-    {
-      "epoch": 0.87,
-      "learning_rate": 4.262792828314637e-05,
-      "loss": 0.1567,
-      "step": 200
-    },
-    {
-      "epoch": 0.9,
-      "learning_rate": 4.282659417003183e-05,
-      "loss": 0.1423,
-      "step": 205
-    },
-    {
-      "epoch": 0.92,
-      "learning_rate": 4.302047245807045e-05,
-      "loss": 0.1405,
-      "step": 210
-    },
-    {
-      "epoch": 0.94,
-      "learning_rate": 4.320978847798302e-05,
-      "loss": 0.1406,
-      "step": 215
-    },
-    {
-      "epoch": 0.96,
-      "learning_rate": 4.3394752017602966e-05,
-      "loss": 0.1381,
-      "step": 220
-    },
-    {
-      "epoch": 0.98,
-      "learning_rate": 4.357555871929799e-05,
-      "loss": 0.131,
-      "step": 225
-    },
-    {
-      "epoch": 1.01,
-      "learning_rate": 4.375239132378962e-05,
-      "loss": 0.1507,
-      "step": 230
-    },
-    {
-      "epoch": 1.03,
-      "learning_rate": 4.392542078019592e-05,
-      "loss": 0.1199,
-      "step": 235
-    },
-    {
-      "epoch": 1.05,
-      "learning_rate": 4.4094807239200284e-05,
-      "loss": 0.1257,
-      "step": 240
-    },
-    {
-      "epoch": 1.07,
-      "learning_rate": 4.426070094380871e-05,
-      "loss": 0.1185,
-      "step": 245
-    },
-    {
-      "epoch": 1.1,
-      "learning_rate": 4.442324303011218e-05,
-      "loss": 0.115,
-      "step": 250
-    },
-    {
-      "epoch": 1.12,
-      "learning_rate": 4.458256624874931e-05,
-      "loss": 0.1188,
-      "step": 255
-    },
-    {
-      "epoch": 1.14,
-      "learning_rate": 4.4738795616310163e-05,
-      "loss": 0.108,
-      "step": 260
-    },
-    {
-      "epoch": 1.16,
-      "learning_rate": 4.48920490046898e-05,
-      "loss": 0.1125,
-      "step": 265
-    },
-    {
-      "epoch": 1.18,
-      "learning_rate": 4.50424376753519e-05,
-      "loss": 0.1094,
-      "step": 270
-    },
-    {
-      "epoch": 1.21,
-      "learning_rate": 4.5190066764568774e-05,
-      "loss": 0.1033,
-      "step": 275
-    },
-    {
-      "epoch": 1.23,
-      "learning_rate": 4.533503572493855e-05,
-      "loss": 0.1037,
-      "step": 280
-    },
-    {
-      "epoch": 1.25,
-      "learning_rate": 4.547743872782376e-05,
-      "loss": 0.1038,
-      "step": 285
-    },
-    {
-      "epoch": 1.27,
-      "learning_rate": 4.5617365030789545e-05,
-      "loss": 0.1003,
-      "step": 290
-    },
-    {
-      "epoch": 1.29,
-      "learning_rate": 4.575489931363226e-05,
-      "loss": 0.1047,
-      "step": 295
-    },
-    {
-      "epoch": 1.31,
-      "learning_rate": 4.589012198616609e-05,
-      "loss": 0.0961,
-      "step": 300
-    },
-    {
-      "epoch": 1.34,
-      "learning_rate": 4.602310947056923e-05,
-      "loss": 0.0959,
-      "step": 305
-    },
-    {
-      "epoch": 1.36,
-      "learning_rate": 4.615393446077182e-05,
-      "loss": 0.0992,
-      "step": 310
-    },
-    {
-      "epoch": 1.38,
-      "learning_rate": 4.6282666161090166e-05,
-      "loss": 0.0938,
-      "step": 315
-    },
-    {
-      "epoch": 1.4,
-      "learning_rate": 4.640937050606839e-05,
-      "loss": 0.0963,
-      "step": 320
-    },
-    {
-      "epoch": 1.42,
-      "learning_rate": 4.653411036327597e-05,
-      "loss": 0.0943,
-      "step": 325
-    },
-    {
-      "epoch": 1.45,
-      "learning_rate": 4.665694572062267e-05,
-      "loss": 0.0889,
-      "step": 330
-    },
-    {
-      "epoch": 1.47,
-      "learning_rate": 4.677793385958802e-05,
-      "loss": 0.0845,
-      "step": 335
-    },
-    {
-      "epoch": 1.49,
-      "learning_rate": 4.689712951561742e-05,
-      "loss": 0.0864,
-      "step": 340
-    },
-    {
-      "epoch": 1.51,
-      "learning_rate": 4.701458502680934e-05,
-      "loss": 0.0869,
-      "step": 345
-    },
-    {
-      "epoch": 1.53,
-      "learning_rate": 4.713035047190436e-05,
-      "loss": 0.086,
-      "step": 350
-    },
-    {
-      "epoch": 1.55,
-      "learning_rate": 4.7244473798486756e-05,
-      "loss": 0.0863,
-      "step": 355
-    },
-    {
-      "epoch": 1.58,
-      "learning_rate": 4.735700094222001e-05,
-      "loss": 0.0824,
-      "step": 360
-    },
-    {
-      "epoch": 1.6,
-      "learning_rate": 4.746797593785841e-05,
-      "loss": 0.0875,
-      "step": 365
-    },
-    {
-      "epoch": 1.62,
-      "learning_rate": 4.7577441022706095e-05,
-      "loss": 0.0842,
-      "step": 370
-    },
-    {
-      "epoch": 1.64,
-      "learning_rate": 4.76854367331319e-05,
-      "loss": 0.0789,
-      "step": 375
-    },
-    {
-      "epoch": 1.66,
-      "learning_rate": 4.7792001994691866e-05,
-      "loss": 0.0791,
-      "step": 380
-    },
-    {
-      "epoch": 1.69,
-      "learning_rate": 4.7897174206360945e-05,
-      "loss": 0.0828,
-      "step": 385
-    },
-    {
-      "epoch": 1.71,
-      "learning_rate": 4.800098931932989e-05,
-      "loss": 0.077,
-      "step": 390
-    },
-    {
-      "epoch": 1.73,
-      "learning_rate": 4.810348191078279e-05,
-      "loss": 0.0776,
-      "step": 395
-    },
-    {
-      "epoch": 1.75,
-      "learning_rate": 4.82046852530342e-05,
-      "loss": 0.0798,
-      "step": 400
-    },
-    {
-      "epoch": 1.77,
-      "learning_rate": 4.830463137837162e-05,
-      "loss": 0.0762,
-      "step": 405
-    },
-    {
-      "epoch": 1.8,
-      "learning_rate": 4.8403351139919656e-05,
-      "loss": 0.0762,
-      "step": 410
-    },
-    {
-      "epoch": 1.82,
-      "learning_rate": 4.850087426881512e-05,
-      "loss": 0.0799,
-      "step": 415
-    },
-    {
-      "epoch": 1.84,
-      "learning_rate": 4.859722942795827e-05,
-      "loss": 0.0757,
-      "step": 420
-    },
-    {
-      "epoch": 1.86,
-      "learning_rate": 4.8692444262583224e-05,
-      "loss": 0.0733,
-      "step": 425
-    },
-    {
-      "epoch": 1.88,
-      "learning_rate": 4.8786545447870833e-05,
-      "loss": 0.0755,
-      "step": 430
-    },
-    {
-      "epoch": 1.9,
-      "learning_rate": 4.8879558733809264e-05,
-      "loss": 0.0731,
-      "step": 435
-    },
-    {
-      "epoch": 1.93,
-      "learning_rate": 4.897150898749078e-05,
-      "loss": 0.0729,
-      "step": 440
-    },
-    {
-      "epoch": 1.95,
-      "learning_rate": 4.9062420233018894e-05,
-      "loss": 0.0738,
-      "step": 445
-    },
-    {
-      "epoch": 1.97,
-      "learning_rate": 4.915231568918581e-05,
-      "loss": 0.0734,
-      "step": 450
-    },
-    {
-      "epoch": 1.99,
-      "learning_rate": 4.924121780506815e-05,
-      "loss": 0.0729,
-      "step": 455
-    },
-    {
-      "epoch": 2.02,
-      "learning_rate": 4.9346619656948594e-05,
-      "loss": 0.082,
-      "step": 460
-    },
-    {
-      "epoch": 2.04,
-      "learning_rate": 4.9433411864583826e-05,
-      "loss": 0.0661,
-      "step": 465
-    },
-    {
-      "epoch": 2.06,
-      "learning_rate": 4.9519277776977253e-05,
-      "loss": 0.0636,
-      "step": 470
-    },
-    {
-      "epoch": 2.08,
-      "learning_rate": 4.9604236957409594e-05,
-      "loss": 0.0666,
-      "step": 475
-    },
-    {
-      "epoch": 2.1,
-      "learning_rate": 4.9688308355869887e-05,
-      "loss": 0.0667,
-      "step": 480
-    },
-    {
-      "epoch": 2.13,
-      "learning_rate": 4.977151033442553e-05,
-      "loss": 0.064,
-      "step": 485
-    },
-    {
-      "epoch": 2.15,
-      "learning_rate": 4.9853860691293774e-05,
-      "loss": 0.0659,
-      "step": 490
-    },
-    {
-      "epoch": 2.17,
-      "learning_rate": 4.993537668369384e-05,
-      "loss": 0.0629,
-      "step": 495
-    },
-    {
-      "epoch": 2.19,
-      "learning_rate": 5e-05,
-      "loss": 0.0665,
-      "step": 500
-    },
-    {
-      "epoch": 2.21,
-      "learning_rate": 5e-05,
-      "loss": 0.0626,
-      "step": 505
-    },
-    {
-      "epoch": 2.24,
-      "learning_rate": 5e-05,
-      "loss": 0.0686,
-      "step": 510
-    },
-    {
-      "epoch": 2.26,
-      "learning_rate": 5e-05,
-      "loss": 0.0652,
-      "step": 515
-    },
-    {
-      "epoch": 2.28,
-      "learning_rate": 5e-05,
-      "loss": 0.0622,
-      "step": 520
-    },
-    {
-      "epoch": 2.3,
-      "learning_rate": 5e-05,
-      "loss": 0.0626,
-      "step": 525
-    },
-    {
-      "epoch": 2.32,
-      "learning_rate": 5e-05,
-      "loss": 0.0658,
-      "step": 530
-    },
-    {
-      "epoch": 2.35,
-      "learning_rate": 5e-05,
-      "loss": 0.0614,
-      "step": 535
-    },
-    {
-      "epoch": 2.37,
-      "learning_rate": 5e-05,
-      "loss": 0.0615,
-      "step": 540
-    },
-    {
-      "epoch": 2.39,
-      "learning_rate": 5e-05,
-      "loss": 0.0626,
-      "step": 545
-    },
-    {
-      "epoch": 2.41,
-      "learning_rate": 5e-05,
-      "loss": 0.0643,
-      "step": 550
-    },
-    {
-      "epoch": 2.43,
-      "learning_rate": 5e-05,
-      "loss": 0.0622,
-      "step": 555
-    },
-    {
-      "epoch": 2.45,
-      "learning_rate": 5e-05,
-      "loss": 0.0645,
-      "step": 560
-    },
-    {
-      "epoch": 2.48,
-      "learning_rate": 5e-05,
-      "loss": 0.0632,
-      "step": 565
-    },
-    {
-      "epoch": 2.5,
-      "learning_rate": 5e-05,
-      "loss": 0.0641,
-      "step": 570
-    },
-    {
-      "epoch": 2.52,
-      "learning_rate": 5e-05,
-      "loss": 0.0607,
-      "step": 575
-    },
-    {
-      "epoch": 2.54,
-      "learning_rate": 5e-05,
-      "loss": 0.0622,
-      "step": 580
-    },
-    {
-      "epoch": 2.56,
-      "learning_rate": 5e-05,
-      "loss": 0.0635,
-      "step": 585
-    },
-    {
-      "epoch": 2.59,
-      "learning_rate": 5e-05,
-      "loss": 0.0619,
-      "step": 590
-    },
-    {
-      "epoch": 2.61,
-      "learning_rate": 5e-05,
-      "loss": 0.0613,
-      "step": 595
-    },
-    {
-      "epoch": 2.63,
-      "learning_rate": 5e-05,
-      "loss": 0.0628,
-      "step": 600
-    },
-    {
-      "epoch": 2.65,
-      "learning_rate": 5e-05,
-      "loss": 0.0613,
-      "step": 605
-    },
-    {
-      "epoch": 2.67,
-      "learning_rate": 5e-05,
-      "loss": 0.064,
-      "step": 610
-    },
-    {
-      "epoch": 2.69,
-      "learning_rate": 5e-05,
-      "loss": 0.0654,
-      "step": 615
-    },
-    {
-      "epoch": 2.72,
-      "learning_rate": 5e-05,
-      "loss": 0.0635,
-      "step": 620
-    },
-    {
-      "epoch": 2.74,
-      "learning_rate": 5e-05,
-      "loss": 0.0611,
-      "step": 625
-    },
-    {
-      "epoch": 2.76,
-      "learning_rate": 5e-05,
-      "loss": 0.0605,
-      "step": 630
-    },
-    {
-      "epoch": 2.78,
-      "learning_rate": 5e-05,
-      "loss": 0.0606,
-      "step": 635
-    },
-    {
-      "epoch": 2.8,
-      "learning_rate": 5e-05,
-      "loss": 0.0647,
-      "step": 640
-    },
-    {
-      "epoch": 2.83,
-      "learning_rate": 5e-05,
-      "loss": 0.0625,
-      "step": 645
-    },
-    {
-      "epoch": 2.85,
-      "learning_rate": 5e-05,
-      "loss": 0.0621,
-      "step": 650
-    },
-    {
-      "epoch": 2.87,
-      "learning_rate": 5e-05,
-      "loss": 0.0607,
-      "step": 655
-    },
-    {
-      "epoch": 2.89,
-      "learning_rate": 5e-05,
-      "loss": 0.0581,
-      "step": 660
-    },
-    {
-      "epoch": 2.91,
-      "learning_rate": 5e-05,
-      "loss": 0.063,
-      "step": 665
-    },
-    {
-      "epoch": 2.94,
-      "learning_rate": 5e-05,
-      "loss": 0.0582,
-      "step": 670
-    },
-    {
-      "epoch": 2.96,
-      "learning_rate": 5e-05,
-      "loss": 0.0598,
-      "step": 675
-    },
-    {
-      "epoch": 2.98,
-      "learning_rate": 5e-05,
-      "loss": 0.0609,
-      "step": 680
-    },
-    {
-      "epoch": 3.0,
-      "learning_rate": 5e-05,
-      "loss": 0.073,
-      "step": 685
-    },
-    {
-      "epoch": 3.03,
-      "learning_rate": 5e-05,
-      "loss": 0.055,
-      "step": 690
-    },
-    {
-      "epoch": 3.05,
-      "learning_rate": 5e-05,
-      "loss": 0.0587,
-      "step": 695
-    },
-    {
-      "epoch": 3.07,
-      "learning_rate": 5e-05,
-      "loss": 0.055,
-      "step": 700
-    },
-    {
-      "epoch": 3.09,
-      "learning_rate": 5e-05,
-      "loss": 0.0559,
-      "step": 705
-    },
-    {
-      "epoch": 3.11,
-      "learning_rate": 5e-05,
-      "loss": 0.0578,
-      "step": 710
-    },
-    {
-      "epoch": 3.14,
-      "learning_rate": 5e-05,
-      "loss": 0.0541,
-      "step": 715
-    },
-    {
-      "epoch": 3.16,
-      "learning_rate": 5e-05,
-      "loss": 0.0558,
-      "step": 720
-    },
-    {
-      "epoch": 3.18,
-      "learning_rate": 5e-05,
-      "loss": 0.0561,
-      "step": 725
-    },
-    {
-      "epoch": 3.2,
-      "learning_rate": 5e-05,
-      "loss": 0.0559,
-      "step": 730
-    },
-    {
-      "epoch": 3.22,
-      "learning_rate": 5e-05,
-      "loss": 0.0544,
-      "step": 735
-    },
-    {
-      "epoch": 3.24,
-      "learning_rate": 5e-05,
-      "loss": 0.0556,
-      "step": 740
-    },
-    {
-      "epoch": 3.27,
-      "learning_rate": 5e-05,
-      "loss": 0.0566,
-      "step": 745
-    },
-    {
-      "epoch": 3.29,
-      "learning_rate": 5e-05,
-      "loss": 0.0551,
-      "step": 750
-    },
-    {
-      "epoch": 3.31,
-      "learning_rate": 5e-05,
-      "loss": 0.0572,
-      "step": 755
-    },
-    {
-      "epoch": 3.33,
-      "learning_rate": 5e-05,
-      "loss": 0.0551,
-      "step": 760
-    },
-    {
-      "epoch": 3.35,
-      "learning_rate": 5e-05,
-      "loss": 0.056,
-      "step": 765
-    },
-    {
-      "epoch": 3.38,
-      "learning_rate": 5e-05,
-      "loss": 0.0547,
-      "step": 770
-    },
-    {
-      "epoch": 3.4,
-      "learning_rate": 5e-05,
-      "loss": 0.0564,
-      "step": 775
-    },
-    {
-      "epoch": 3.42,
-      "learning_rate": 5e-05,
-      "loss": 0.0569,
-      "step": 780
-    },
-    {
-      "epoch": 3.44,
-      "learning_rate": 5e-05,
-      "loss": 0.0559,
-      "step": 785
-    },
-    {
-      "epoch": 3.46,
-      "learning_rate": 5e-05,
-      "loss": 0.0548,
-      "step": 790
-    },
-    {
-      "epoch": 3.48,
-      "learning_rate": 5e-05,
-      "loss": 0.0548,
-      "step": 795
-    },
-    {
-      "epoch": 3.51,
-      "learning_rate": 5e-05,
-      "loss": 0.0537,
-      "step": 800
-    },
-    {
-      "epoch": 3.53,
-      "learning_rate": 5e-05,
-      "loss": 0.0547,
-      "step": 805
-    },
-    {
-      "epoch": 3.55,
-      "learning_rate": 5e-05,
-      "loss": 0.0559,
-      "step": 810
-    },
-    {
-      "epoch": 3.57,
-      "learning_rate": 5e-05,
-      "loss": 0.057,
-      "step": 815
-    },
-    {
-      "epoch": 3.59,
-      "learning_rate": 5e-05,
-      "loss": 0.0561,
-      "step": 820
-    },
-    {
-      "epoch": 3.62,
-      "learning_rate": 5e-05,
-      "loss": 0.0546,
-      "step": 825
-    },
-    {
-      "epoch": 3.64,
-      "learning_rate": 5e-05,
-      "loss": 0.0552,
-      "step": 830
-    },
-    {
-      "epoch": 3.66,
-      "learning_rate": 5e-05,
-      "loss": 0.0566,
-      "step": 835
-    },
-    {
-      "epoch": 3.68,
-      "learning_rate": 5e-05,
-      "loss": 0.0531,
-      "step": 840
-    },
-    {
-      "epoch": 3.7,
-      "learning_rate": 5e-05,
-      "loss": 0.056,
-      "step": 845
-    },
-    {
-      "epoch": 3.73,
-      "learning_rate": 5e-05,
-      "loss": 0.0557,
-      "step": 850
-    },
-    {
-      "epoch": 3.75,
-      "learning_rate": 5e-05,
-      "loss": 0.057,
-      "step": 855
-    },
-    {
-      "epoch": 3.77,
-      "learning_rate": 5e-05,
-      "loss": 0.0539,
-      "step": 860
-    },
-    {
-      "epoch": 3.79,
-      "learning_rate": 5e-05,
-      "loss": 0.0529,
-      "step": 865
-    },
-    {
-      "epoch": 3.81,
-      "learning_rate": 5e-05,
-      "loss": 0.0552,
-      "step": 870
-    },
-    {
-      "epoch": 3.83,
-      "learning_rate": 5e-05,
-      "loss": 0.0547,
-      "step": 875
-    },
-    {
-      "epoch": 3.86,
-      "learning_rate": 5e-05,
-      "loss": 0.0553,
-      "step": 880
-    },
-    {
-      "epoch": 3.88,
-      "learning_rate": 5e-05,
-      "loss": 0.0558,
-      "step": 885
-    },
-    {
-      "epoch": 3.9,
-      "learning_rate": 5e-05,
-      "loss": 0.054,
-      "step": 890
-    },
-    {
-      "epoch": 3.92,
-      "learning_rate": 5e-05,
-      "loss": 0.0549,
-      "step": 895
-    },
-    {
-      "epoch": 3.94,
-      "learning_rate": 5e-05,
-      "loss": 0.0544,
-      "step": 900
-    },
-    {
-      "epoch": 3.97,
-      "learning_rate": 5e-05,
-      "loss": 0.0558,
-      "step": 905
-    },
-    {
-      "epoch": 3.99,
-      "learning_rate": 5e-05,
-      "loss": 0.0545,
-      "step": 910
-    },
-    {
-      "epoch": 4.01,
-      "learning_rate": 5e-05,
-      "loss": 0.0604,
-      "step": 915
-    },
-    {
-      "epoch": 4.03,
-      "learning_rate": 5e-05,
-      "loss": 0.0497,
-      "step": 920
-    },
-    {
-      "epoch": 4.06,
-      "learning_rate": 5e-05,
-      "loss": 0.049,
-      "step": 925
-    },
-    {
-      "epoch": 4.08,
-      "learning_rate": 5e-05,
-      "loss": 0.0488,
-      "step": 930
-    },
-    {
-      "epoch": 4.1,
-      "learning_rate": 5e-05,
-      "loss": 0.0495,
-      "step": 935
-    },
-    {
-      "epoch": 4.12,
-      "learning_rate": 5e-05,
-      "loss": 0.049,
-      "step": 940
-    },
-    {
-      "epoch": 4.14,
-      "learning_rate": 5e-05,
-      "loss": 0.0502,
-      "step": 945
-    },
-    {
-      "epoch": 4.17,
-      "learning_rate": 5e-05,
-      "loss": 0.0493,
-      "step": 950
-    },
-    {
-      "epoch": 4.19,
-      "learning_rate": 5e-05,
-      "loss": 0.0496,
-      "step": 955
-    },
-    {
-      "epoch": 4.21,
-      "learning_rate": 5e-05,
-      "loss": 0.0475,
-      "step": 960
-    },
-    {
-      "epoch": 4.23,
-      "learning_rate": 5e-05,
-      "loss": 0.0486,
-      "step": 965
-    },
-    {
-      "epoch": 4.25,
-      "learning_rate": 5e-05,
-      "loss": 0.0503,
-      "step": 970
-    },
-    {
-      "epoch": 4.28,
-      "learning_rate": 5e-05,
-      "loss": 0.0508,
-      "step": 975
-    },
-    {
-      "epoch": 4.3,
-      "learning_rate": 5e-05,
-      "loss": 0.0501,
-      "step": 980
-    },
-    {
-      "epoch": 4.32,
-      "learning_rate": 5e-05,
-      "loss": 0.0499,
-      "step": 985
-    },
-    {
-      "epoch": 4.34,
-      "learning_rate": 5e-05,
-      "loss": 0.0485,
-      "step": 990
-    },
-    {
-      "epoch": 4.36,
-      "learning_rate": 5e-05,
-      "loss": 0.0494,
-      "step": 995
-    },
-    {
-      "epoch": 4.38,
-      "learning_rate": 5e-05,
-      "loss": 0.0503,
-      "step": 1000
-    },
-    {
-      "epoch": 4.41,
-      "learning_rate": 5e-05,
-      "loss": 0.0512,
-      "step": 1005
-    },
-    {
-      "epoch": 4.43,
-      "learning_rate": 5e-05,
-      "loss": 0.0513,
-      "step": 1010
-    },
-    {
-      "epoch": 4.45,
-      "learning_rate": 5e-05,
-      "loss": 0.0496,
-      "step": 1015
-    },
-    {
-      "epoch": 4.47,
-      "learning_rate": 5e-05,
-      "loss": 0.0493,
-      "step": 1020
-    },
-    {
-      "epoch": 4.49,
-      "learning_rate": 5e-05,
-      "loss": 0.0516,
-      "step": 1025
-    },
-    {
-      "epoch": 4.52,
-      "learning_rate": 5e-05,
-      "loss": 0.0498,
-      "step": 1030
-    },
-    {
-      "epoch": 4.54,
-      "learning_rate": 5e-05,
-      "loss": 0.0498,
-      "step": 1035
-    },
-    {
-      "epoch": 4.56,
-      "learning_rate": 5e-05,
-      "loss": 0.0491,
-      "step": 1040
-    },
-    {
-      "epoch": 4.58,
-      "learning_rate": 5e-05,
-      "loss": 0.047,
-      "step": 1045
-    },
-    {
-      "epoch": 4.6,
-      "learning_rate": 5e-05,
-      "loss": 0.0493,
-      "step": 1050
-    },
-    {
-      "epoch": 4.62,
-      "learning_rate": 5e-05,
-      "loss": 0.0488,
-      "step": 1055
-    },
-    {
-      "epoch": 4.65,
-      "learning_rate": 5e-05,
-      "loss": 0.0502,
-      "step": 1060
-    },
-    {
-      "epoch": 4.67,
-      "learning_rate": 5e-05,
-      "loss": 0.0511,
-      "step": 1065
-    },
-    {
-      "epoch": 4.69,
-      "learning_rate": 5e-05,
-      "loss": 0.0498,
-      "step": 1070
-    },
-    {
-      "epoch": 4.71,
-      "learning_rate": 5e-05,
-      "loss": 0.0511,
-      "step": 1075
-    },
-    {
-      "epoch": 4.73,
-      "learning_rate": 5e-05,
-      "loss": 0.0498,
-      "step": 1080
-    },
-    {
-      "epoch": 4.76,
-      "learning_rate": 5e-05,
-      "loss": 0.0521,
-      "step": 1085
-    },
-    {
-      "epoch": 4.78,
-      "learning_rate": 5e-05,
-      "loss": 0.0503,
-      "step": 1090
-    },
-    {
-      "epoch": 4.8,
-      "learning_rate": 5e-05,
-      "loss": 0.0509,
-      "step": 1095
-    },
-    {
-      "epoch": 4.82,
-      "learning_rate": 5e-05,
-      "loss": 0.0523,
-      "step": 1100
-    },
-    {
-      "epoch": 4.84,
-      "learning_rate": 5e-05,
-      "loss": 0.0465,
-      "step": 1105
-    },
-    {
-      "epoch": 4.87,
-      "learning_rate": 5e-05,
-      "loss": 0.0521,
-      "step": 1110
-    },
-    {
-      "epoch": 4.89,
-      "learning_rate": 5e-05,
-      "loss": 0.0488,
-      "step": 1115
-    },
-    {
-      "epoch": 4.91,
-      "learning_rate": 5e-05,
-      "loss": 0.0488,
-      "step": 1120
-    },
-    {
-      "epoch": 4.93,
-      "learning_rate": 5e-05,
-      "loss": 0.0502,
-      "step": 1125
-    },
-    {
-      "epoch": 4.95,
-      "learning_rate": 5e-05,
-      "loss": 0.048,
-      "step": 1130
-    },
-    {
-      "epoch": 4.97,
-      "learning_rate": 5e-05,
-      "loss": 0.0497,
-      "step": 1135
-    },
-    {
-      "epoch": 5.0,
-      "learning_rate": 5e-05,
-      "loss": 0.0484,
-      "step": 1140
-    }
-  ],
-  "max_steps": 1140,
-  "num_train_epochs": 5,
-  "total_flos": 8.693964764402942e+18,
-  "trial_name": null,
-  "trial_params": null
-}

last-checkpoint/training_args.bin DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:35e1a0fcbf25720ed542bd10424c03d13a02ac8f5eff85f51e6389fe1f87d97f
-size 4603

last-checkpoint/vocab.json DELETED Viewed

The diff for this file is too large to render. See raw diff

last-checkpoint/zero_to_fp32.py DELETED Viewed

@@ -1,482 +0,0 @@
-#!/usr/bin/env python
-# This script extracts fp32 consolidated weights from a zero 2 and 3 DeepSpeed checkpoints. It gets
-# copied into the top level checkpoint dir, so the user can easily do the conversion at any point in
-# the future. Once extracted, the weights don't require DeepSpeed and can be used in any
-# application.
-#
-# example: python zero_to_fp32.py . pytorch_model.bin
-import argparse
-import torch
-import glob
-import math
-import os
-import re
-from collections import OrderedDict
-# while this script doesn't use deepspeed to recover data, since the checkpoints are pickled with
-# DeepSpeed data structures it has to be available in the current python environment.
-from deepspeed.utils import logger
-from deepspeed.checkpoint.constants import (DS_VERSION,
-                                            OPTIMIZER_STATE_DICT,
-                                            SINGLE_PARTITION_OF_FP32_GROUPS,
-                                            FP32_FLAT_GROUPS,
-                                            ZERO_STAGE,
-                                            PARTITION_COUNT,
-                                            PARAM_SHAPES,
-                                            BUFFER_NAMES)
-debug = 0
-# load to cpu
-device = torch.device('cpu')
-def atoi(text):
-    return int(text) if text.isdigit() else text
-def natural_keys(text):
-    '''
-    alist.sort(key=natural_keys) sorts in human order
-    http://nedbatchelder.com/blog/200712/human_sorting.html
-    (See Toothy's implementation in the comments)
-    '''
-    return [atoi(c) for c in re.split(r'(\d+)', text)]
-def get_model_state_file(checkpoint_dir, zero_stage):
-    if not os.path.isdir(checkpoint_dir):
-        raise FileNotFoundError(f"Directory '{checkpoint_dir}' doesn't exist")
-    # there should be only one file
-    if zero_stage == 2:
-        file = os.path.join(checkpoint_dir, "mp_rank_00_model_states.pt")
-    elif zero_stage == 3:
-        file = os.path.join(checkpoint_dir, "zero_pp_rank_0_mp_rank_00_model_states.pt")
-    if not os.path.exists(file):
-        raise FileNotFoundError(f"can't find model states file at '{file}'")
-    return file
-def get_optim_files(checkpoint_dir):
-    # XXX: need to test that this simple glob rule works for multi-node setup too
-    optim_files = sorted(glob.glob(os.path.join(checkpoint_dir,
-                                                "*_optim_states.pt")),
-                         key=natural_keys)
-    if len(optim_files) == 0:
-        raise FileNotFoundError(
-            f"can't find '*_optim_states.pt' files in directory '{checkpoint_dir}'")
-    return optim_files
-def parse_model_state(file):
-    state_dict = torch.load(file, map_location=device)
-    if BUFFER_NAMES not in state_dict:
-        raise ValueError(f"{file} is not a model state checkpoint")
-    buffer_names = state_dict[BUFFER_NAMES]
-    if debug:
-        print("Found buffers:", buffer_names)
-    # recover just the buffers while restoring them to fp32 if they were saved in fp16
-    buffers = {
-        k: v.float()
-        for k,
-        v in state_dict["module"].items() if k in buffer_names
-    }
-    param_shapes = state_dict[PARAM_SHAPES]
-    ds_version = state_dict.get(DS_VERSION, None)
-    return buffers, param_shapes, ds_version
-def parse_optim_states(files, ds_checkpoint_dir):
-    total_files = len(files)
-    state_dicts = []
-    for f in files:
-        state_dicts.append(torch.load(f, map_location=device))
-    if not ZERO_STAGE in state_dicts[0][OPTIMIZER_STATE_DICT]:
-        raise ValueError(f"{files[0]} is not a zero checkpoint")
-    zero_stage = state_dicts[0][OPTIMIZER_STATE_DICT][ZERO_STAGE]
-    world_size = state_dicts[0][OPTIMIZER_STATE_DICT][PARTITION_COUNT]
-    # For ZeRO-2 each param group can have different partition_count as data parallelism for expert
-    # parameters can be different from data parallelism for non-expert parameters. So we can just
-    # use the max of the partition_count to get the dp world_size.
-    if type(world_size) is list:
-        world_size = max(world_size)
-    if world_size != total_files:
-        raise ValueError(
-            f"Expected {world_size} of '*_optim_states.pt' under '{ds_checkpoint_dir}' but found {total_files} files. "
-            "Possibly due to an overwrite of an old checkpoint, or a checkpoint didn't get saved by one or more processes."
-        )
-    # the groups are named differently in each stage
-    if zero_stage == 2:
-        fp32_groups_key = SINGLE_PARTITION_OF_FP32_GROUPS
-    elif zero_stage == 3:
-        fp32_groups_key = FP32_FLAT_GROUPS
-    else:
-        raise ValueError(f"unknown zero stage {zero_stage}")
-    if zero_stage == 2:
-        fp32_flat_groups = [
-            state_dicts[i][OPTIMIZER_STATE_DICT][fp32_groups_key]
-            for i in range(len(state_dicts))
-        ]
-    elif zero_stage == 3:
-        # if there is more than one param group, there will be multiple flattened tensors - one
-        # flattened tensor per group - for simplicity merge them into a single tensor
-        #
-        # XXX: could make the script more memory efficient for when there are multiple groups - it
-        # will require matching the sub-lists of param_shapes for each param group flattened tensor
-        fp32_flat_groups = [
-            torch.cat(state_dicts[i][OPTIMIZER_STATE_DICT][fp32_groups_key],
-                      0) for i in range(len(state_dicts))
-        ]
-    return zero_stage, world_size, fp32_flat_groups
-def _get_fp32_state_dict_from_zero_checkpoint(ds_checkpoint_dir):
-    """
-    Returns fp32 state_dict reconstructed from ds checkpoint
-    Args:
-        - ``ds_checkpoint_dir``: path to the deepspeed checkpoint folder (where the optimizer files are)
-    """
-    print(f"Processing zero checkpoint '{ds_checkpoint_dir}'")
-    optim_files = get_optim_files(ds_checkpoint_dir)
-    zero_stage, world_size, fp32_flat_groups = parse_optim_states(optim_files, ds_checkpoint_dir)
-    print(
-        f"Detected checkpoint of type zero stage {zero_stage}, world_size: {world_size}")
-    model_file = get_model_state_file(ds_checkpoint_dir, zero_stage)
-    buffers, param_shapes, ds_version = parse_model_state(model_file)
-    print(f'Parsing checkpoint created by deepspeed=={ds_version}')
-    if zero_stage == 2:
-        return _get_fp32_state_dict_from_zero2_checkpoint(world_size,
-                                                          param_shapes,
-                                                          fp32_flat_groups,
-                                                          buffers)
-    elif zero_stage == 3:
-        return _get_fp32_state_dict_from_zero3_checkpoint(world_size,
-                                                          param_shapes,
-                                                          fp32_flat_groups,
-                                                          buffers)
-def _get_fp32_state_dict_from_zero2_checkpoint(world_size,
-                                               param_shapes,
-                                               fp32_flat_groups,
-                                               buffers):
-    # Reconstruction protocol:
-    #
-    # XXX: document this
-    if debug:
-        for i in range(world_size):
-            for j in range(len(fp32_flat_groups[0])):
-                print(
-                    f"{FP32_FLAT_GROUPS}[{i}][{j}].shape={fp32_flat_groups[i][j].shape}")
-    # XXX: memory usage doubles here (zero2)
-    num_param_groups = len(fp32_flat_groups[0])
-    merged_single_partition_of_fp32_groups = []
-    for i in range(num_param_groups):
-        merged_partitions = [sd[i] for sd in fp32_flat_groups]
-        full_single_fp32_vector = torch.cat(merged_partitions, 0)
-        merged_single_partition_of_fp32_groups.append(full_single_fp32_vector)
-    avail_numel = sum([
-        full_single_fp32_vector.numel()
-        for full_single_fp32_vector in merged_single_partition_of_fp32_groups
-    ])
-    if debug:
-        wanted_params = sum([len(shapes) for shapes in param_shapes])
-        wanted_numel = sum(
-            [sum(shape.numel() for shape in shapes.values()) for shapes in param_shapes])
-        # not asserting if there is a mismatch due to possible padding
-        print(f"Have {avail_numel} numels to process.")
-        print(f"Need {wanted_numel} numels in {wanted_params} params.")
-    state_dict = OrderedDict()
-    # buffers
-    state_dict.update(buffers)
-    if debug:
-        print(f"added {len(buffers)} buffers")
-    # params
-    # XXX: for huge models that can't fit into the host's RAM we will have to recode this to support
-    # out-of-core computing solution
-    total_numel = 0
-    total_params = 0
-    for shapes, full_single_fp32_vector in zip(param_shapes, merged_single_partition_of_fp32_groups):
-        offset = 0
-        avail_numel = full_single_fp32_vector.numel()
-        for name, shape in shapes.items():
-            unpartitioned_numel = shape.numel()
-            total_numel += unpartitioned_numel
-            total_params += 1
-            if debug:
-                print(
-                    f"{name} full shape: {shape} unpartitioned numel {unpartitioned_numel} "
-                )
-            state_dict[name] = full_single_fp32_vector.narrow(
-                0,
-                offset,
-                unpartitioned_numel).view(shape)
-            offset += unpartitioned_numel
-        # Z2 started to align to 2*world_size to improve nccl performance. Therefore both offset and
-        # avail_numel can differ by anywhere between 0..2*world_size. Due to two unrelated complex
-        # paddings performed in the code it's almost impossible to predict the exact numbers w/o the
-        # live optimizer object, so we are checking that the numbers are within the right range
-        align_to = 2 * world_size
-        def zero2_align(x):
-            return align_to * math.ceil(x / align_to)
-        if debug:
-            print(f"original offset={offset}, avail_numel={avail_numel}")
-        offset = zero2_align(offset)
-        avail_numel = zero2_align(avail_numel)
-        if debug:
-            print(f"aligned  offset={offset}, avail_numel={avail_numel}")
-        # Sanity check
-        if offset != avail_numel:
-            raise ValueError(
-                f"consumed {offset} numels out of {avail_numel} - something is wrong")
-    print(
-        f"Reconstructed fp32 state dict with {total_params} params {total_numel} elements"
-    )
-    return state_dict
-def zero3_partitioned_param_info(unpartitioned_numel, world_size):
-    remainder = unpartitioned_numel % world_size
-    padding_numel = (world_size - remainder) if remainder else 0
-    partitioned_numel = math.ceil(unpartitioned_numel / world_size)
-    return partitioned_numel, padding_numel
-def _get_fp32_state_dict_from_zero3_checkpoint(world_size,
-                                               param_shapes,
-                                               fp32_flat_groups,
-                                               buffers):
-    # Reconstruction protocol: For zero3 we need to zip the partitions together at boundary of each
-    # param, re-consolidating each param, while dealing with padding if any
-    avail_numel = fp32_flat_groups[0].numel() * world_size
-    # merge list of dicts, preserving order
-    param_shapes = {k: v for d in param_shapes for k, v in d.items()}
-    if debug:
-        for i in range(world_size):
-            print(f"{FP32_FLAT_GROUPS}[{i}].shape={fp32_flat_groups[i].shape}")
-        wanted_params = len(param_shapes)
-        wanted_numel = sum(shape.numel() for shape in param_shapes.values())
-        # not asserting if there is a mismatch due to possible padding
-        print(f"Have {avail_numel} numels to process.")
-        print(f"Need {wanted_numel} numels in {wanted_params} params.")
-    state_dict = OrderedDict()
-    # buffers
-    state_dict.update(buffers)
-    if debug:
-        print(f"added {len(buffers)} buffers")
-    # params
-    # XXX: for huge models that can't fit into the host's RAM we will have to recode this to support
-    # out-of-core computing solution
-    offset = 0
-    total_numel = 0
-    total_params = 0
-    for name, shape in param_shapes.items():
-        unpartitioned_numel = shape.numel()
-        total_numel += unpartitioned_numel
-        total_params += 1
-        partitioned_numel, partitioned_padding_numel = zero3_partitioned_param_info(unpartitioned_numel, world_size)
-        if debug:
-            print(
-                f"{total_params} {name} full shape: {shape} partition0 numel={partitioned_numel} partitioned_padding_numel={partitioned_padding_numel}"
-            )
-        # XXX: memory usage doubles here
-        state_dict[name] = torch.cat(
-            tuple(fp32_flat_groups[i].narrow(0,
-                                             offset,
-                                             partitioned_numel)
-                  for i in range(world_size)),
-            0).narrow(0,
-                      0,
-                      unpartitioned_numel).view(shape)
-        offset += partitioned_numel
-    offset *= world_size
-    # Sanity check
-    if offset != avail_numel:
-        raise ValueError(
-            f"consumed {offset} numels out of {avail_numel} - something is wrong")
-    print(
-        f"Reconstructed fp32 state dict with {total_params} params {total_numel} elements"
-    )
-    return state_dict
-def get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag=None):
-    """
-    Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated state_dict that can be loaded with
-    ``load_state_dict()`` and used for training without DeepSpeed or shared with others, for example
-    via a model hub.
-    Args:
-        - ``checkpoint_dir``: path to the desired checkpoint folder
-        - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in 'latest' file. e.g., ``global_step14``
-    Returns:
-        - pytorch ``state_dict``
-    Note: this approach may not work if your application doesn't have sufficient free CPU memory and
-    you may need to use the offline approach using the ``zero_to_fp32.py`` script that is saved with
-    the checkpoint.
-    A typical usage might be ::
-        from deepspeed.utils.zero_to_fp32 import get_fp32_state_dict_from_zero_checkpoint
-        # do the training and checkpoint saving
-        state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir) # already on cpu
-        model = model.cpu() # move to cpu
-        model.load_state_dict(state_dict)
-        # submit to model hub or save the model to share with others
-    In this example the ``model`` will no longer be usable in the deepspeed context of the same
-    application. i.e. you will need to re-initialize the deepspeed engine, since
-    ``model.load_state_dict(state_dict)`` will remove all the deepspeed magic from it.
-    If you want it all done for you, use ``load_state_dict_from_zero_checkpoint`` instead.
-    """
-    if tag is None:
-        latest_path = os.path.join(checkpoint_dir, 'latest')
-        if os.path.isfile(latest_path):
-            with open(latest_path, 'r') as fd:
-                tag = fd.read().strip()
-        else:
-            raise ValueError(f"Unable to find 'latest' file at {latest_path}")
-    ds_checkpoint_dir = os.path.join(checkpoint_dir, tag)
-    if not os.path.isdir(ds_checkpoint_dir):
-        raise FileNotFoundError(f"Directory '{ds_checkpoint_dir}' doesn't exist")
-    return _get_fp32_state_dict_from_zero_checkpoint(ds_checkpoint_dir)
-def convert_zero_checkpoint_to_fp32_state_dict(checkpoint_dir, output_file, tag=None):
-    """
-    Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated ``state_dict`` file that can be
-    loaded with ``torch.load(file)`` + ``load_state_dict()`` and used for training without DeepSpeed.
-    Args:
-        - ``checkpoint_dir``: path to the desired checkpoint folder. (one that contains the tag-folder, like ``global_step14``)
-        - ``output_file``: path to the pytorch fp32 state_dict output file (e.g. path/pytorch_model.bin)
-        - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in the file named ``latest`` in the checkpoint folder, e.g., ``global_step14``
-    """
-    state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag)
-    print(f"Saving fp32 state dict to {output_file}")
-    torch.save(state_dict, output_file)
-def load_state_dict_from_zero_checkpoint(model, checkpoint_dir, tag=None):
-    """
-    1. Put the provided model to cpu
-    2. Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated ``state_dict``
-    3. Load it into the provided model
-    Args:
-        - ``model``: the model object to update
-        - ``checkpoint_dir``: path to the desired checkpoint folder. (one that contains the tag-folder, like ``global_step14``)
-        - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in the file named ``latest`` in the checkpoint folder, e.g., ``global_step14``
-    Returns:
-        - ``model`: modified model
-    Make sure you have plenty of CPU memory available before you call this function. If you don't
-    have enough use the ``zero_to_fp32.py`` utility to do the conversion. You will find it
-    conveniently placed for you in the checkpoint folder.
-    A typical usage might be ::
-        from deepspeed.utils.zero_to_fp32 import load_state_dict_from_zero_checkpoint
-        model = load_state_dict_from_zero_checkpoint(trainer.model, checkpoint_dir)
-        # submit to model hub or save the model to share with others
-    Note, that once this was run, the ``model`` will no longer be usable in the deepspeed context
-    of the same application. i.e. you will need to re-initialize the deepspeed engine, since
-    ``model.load_state_dict(state_dict)`` will remove all the deepspeed magic from it.
-    """
-    logger.info(f"Extracting fp32 weights")
-    state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag)
-    logger.info(f"Overwriting model with fp32 weights")
-    model = model.cpu()
-    model.load_state_dict(state_dict, strict=False)
-    return model
-if __name__ == "__main__":
-    parser = argparse.ArgumentParser()
-    parser.add_argument(
-        "checkpoint_dir",
-        type=str,
-        help="path to the desired checkpoint folder, e.g., path/checkpoint-12")
-    parser.add_argument(
-        "output_file",
-        type=str,
-        help=
-        "path to the pytorch fp32 state_dict output file (e.g. path/checkpoint-12/pytorch_model.bin)"
-    )
-    parser.add_argument("-d", "--debug", action='store_true', help="enable debug")
-    args = parser.parse_args()
-    debug = args.debug
-    convert_zero_checkpoint_to_fp32_state_dict(args.checkpoint_dir, args.output_file)