.

Browse files

Files changed (9) hide show

added_tokens.json +945 -0
config.json +48 -0
configuration_codegen.py +220 -0
merges.txt +0 -0
modeling_codegen.py +747 -0
special_tokens_map.json +5 -0
tokenizer.json +0 -0
tokenizer_config.json +10 -0
vocab.json +0 -0

added_tokens.json ADDED Viewed

	@@ -0,0 +1,945 @@

+{
+  "\t\t": 50294,
+  "\t\t\t": 50293,
+  "\t\t\t\t": 50292,
+  "\t\t\t\t\t": 50291,
+  "\t\t\t\t\t\t": 50290,
+  "\t\t\t\t\t\t\t": 50289,
+  "\t\t\t\t\t\t\t\t": 50288,
+  "\t\t\t\t\t\t\t\t\t": 50287,
+  "  ": 50286,
+  "   ": 50285,
+  "    ": 50284,
+  "     ": 50283,
+  "      ": 50282,
+  "       ": 50281,
+  "        ": 50280,
+  "         ": 50279,
+  "          ": 50278,
+  "           ": 50277,
+  "            ": 50276,
+  "             ": 50275,
+  "              ": 50274,
+  "               ": 50273,
+  "                ": 50272,
+  "                 ": 50271,
+  "                  ": 50270,
+  "                   ": 50269,
+  "                    ": 50268,
+  "                     ": 50267,
+  "                      ": 50266,
+  "                       ": 50265,
+  "                        ": 50264,
+  "                         ": 50263,
+  "                          ": 50262,
+  "                           ": 50261,
+  "                            ": 50260,
+  "                             ": 50259,
+  "                              ": 50258,
+  "                               ": 50257,
+  "<dummy_0>": 50295,
+  "<dummy_1>": 50296,
+  "<dummy_2>": 50297,
+  "<dummy_3>": 50298,
+  "<eom>": 50300,
+  "<mask_100>": 51100,
+  "<mask_101>": 51099,
+  "<mask_102>": 51098,
+  "<mask_103>": 51097,
+  "<mask_104>": 51096,
+  "<mask_105>": 51095,
+  "<mask_106>": 51094,
+  "<mask_107>": 51093,
+  "<mask_108>": 51092,
+  "<mask_109>": 51091,
+  "<mask_10>": 51190,
+  "<mask_110>": 51090,
+  "<mask_111>": 51089,
+  "<mask_112>": 51088,
+  "<mask_113>": 51087,
+  "<mask_114>": 51086,
+  "<mask_115>": 51085,
+  "<mask_116>": 51084,
+  "<mask_117>": 51083,
+  "<mask_118>": 51082,
+  "<mask_119>": 51081,
+  "<mask_11>": 51189,
+  "<mask_120>": 51080,
+  "<mask_121>": 51079,
+  "<mask_122>": 51078,
+  "<mask_123>": 51077,
+  "<mask_124>": 51076,
+  "<mask_125>": 51075,
+  "<mask_126>": 51074,
+  "<mask_127>": 51073,
+  "<mask_128>": 51072,
+  "<mask_129>": 51071,
+  "<mask_12>": 51188,
+  "<mask_130>": 51070,
+  "<mask_131>": 51069,
+  "<mask_132>": 51068,
+  "<mask_133>": 51067,
+  "<mask_134>": 51066,
+  "<mask_135>": 51065,
+  "<mask_136>": 51064,
+  "<mask_137>": 51063,
+  "<mask_138>": 51062,
+  "<mask_139>": 51061,
+  "<mask_13>": 51187,
+  "<mask_140>": 51060,
+  "<mask_141>": 51059,
+  "<mask_142>": 51058,
+  "<mask_143>": 51057,
+  "<mask_144>": 51056,
+  "<mask_145>": 51055,
+  "<mask_146>": 51054,
+  "<mask_147>": 51053,
+  "<mask_148>": 51052,
+  "<mask_149>": 51051,
+  "<mask_14>": 51186,
+  "<mask_150>": 51050,
+  "<mask_151>": 51049,
+  "<mask_152>": 51048,
+  "<mask_153>": 51047,
+  "<mask_154>": 51046,
+  "<mask_155>": 51045,
+  "<mask_156>": 51044,
+  "<mask_157>": 51043,
+  "<mask_158>": 51042,
+  "<mask_159>": 51041,
+  "<mask_15>": 51185,
+  "<mask_160>": 51040,
+  "<mask_161>": 51039,
+  "<mask_162>": 51038,
+  "<mask_163>": 51037,
+  "<mask_164>": 51036,
+  "<mask_165>": 51035,
+  "<mask_166>": 51034,
+  "<mask_167>": 51033,
+  "<mask_168>": 51032,
+  "<mask_169>": 51031,
+  "<mask_16>": 51184,
+  "<mask_170>": 51030,
+  "<mask_171>": 51029,
+  "<mask_172>": 51028,
+  "<mask_173>": 51027,
+  "<mask_174>": 51026,
+  "<mask_175>": 51025,
+  "<mask_176>": 51024,
+  "<mask_177>": 51023,
+  "<mask_178>": 51022,
+  "<mask_179>": 51021,
+  "<mask_17>": 51183,
+  "<mask_180>": 51020,
+  "<mask_181>": 51019,
+  "<mask_182>": 51018,
+  "<mask_183>": 51017,
+  "<mask_184>": 51016,
+  "<mask_185>": 51015,
+  "<mask_186>": 51014,
+  "<mask_187>": 51013,
+  "<mask_188>": 51012,
+  "<mask_189>": 51011,
+  "<mask_18>": 51182,
+  "<mask_190>": 51010,
+  "<mask_191>": 51009,
+  "<mask_192>": 51008,
+  "<mask_193>": 51007,
+  "<mask_194>": 51006,
+  "<mask_195>": 51005,
+  "<mask_196>": 51004,
+  "<mask_197>": 51003,
+  "<mask_198>": 51002,
+  "<mask_199>": 51001,
+  "<mask_19>": 51181,
+  "<mask_1>": 51199,
+  "<mask_200>": 51000,
+  "<mask_201>": 50999,
+  "<mask_202>": 50998,
+  "<mask_203>": 50997,
+  "<mask_204>": 50996,
+  "<mask_205>": 50995,
+  "<mask_206>": 50994,
+  "<mask_207>": 50993,
+  "<mask_208>": 50992,
+  "<mask_209>": 50991,
+  "<mask_20>": 51180,
+  "<mask_210>": 50990,
+  "<mask_211>": 50989,
+  "<mask_212>": 50988,
+  "<mask_213>": 50987,
+  "<mask_214>": 50986,
+  "<mask_215>": 50985,
+  "<mask_216>": 50984,
+  "<mask_217>": 50983,
+  "<mask_218>": 50982,
+  "<mask_219>": 50981,
+  "<mask_21>": 51179,
+  "<mask_220>": 50980,
+  "<mask_221>": 50979,
+  "<mask_222>": 50978,
+  "<mask_223>": 50977,
+  "<mask_224>": 50976,
+  "<mask_225>": 50975,
+  "<mask_226>": 50974,
+  "<mask_227>": 50973,
+  "<mask_228>": 50972,
+  "<mask_229>": 50971,
+  "<mask_22>": 51178,
+  "<mask_230>": 50970,
+  "<mask_231>": 50969,
+  "<mask_232>": 50968,
+  "<mask_233>": 50967,
+  "<mask_234>": 50966,
+  "<mask_235>": 50965,
+  "<mask_236>": 50964,
+  "<mask_237>": 50963,
+  "<mask_238>": 50962,
+  "<mask_239>": 50961,
+  "<mask_23>": 51177,
+  "<mask_240>": 50960,
+  "<mask_241>": 50959,
+  "<mask_242>": 50958,
+  "<mask_243>": 50957,
+  "<mask_244>": 50956,
+  "<mask_245>": 50955,
+  "<mask_246>": 50954,
+  "<mask_247>": 50953,
+  "<mask_248>": 50952,
+  "<mask_249>": 50951,
+  "<mask_24>": 51176,
+  "<mask_250>": 50950,
+  "<mask_251>": 50949,
+  "<mask_252>": 50948,
+  "<mask_253>": 50947,
+  "<mask_254>": 50946,
+  "<mask_255>": 50945,
+  "<mask_256>": 50944,
+  "<mask_257>": 50943,
+  "<mask_258>": 50942,
+  "<mask_259>": 50941,
+  "<mask_25>": 51175,
+  "<mask_260>": 50940,
+  "<mask_261>": 50939,
+  "<mask_262>": 50938,
+  "<mask_263>": 50937,
+  "<mask_264>": 50936,
+  "<mask_265>": 50935,
+  "<mask_266>": 50934,
+  "<mask_267>": 50933,
+  "<mask_268>": 50932,
+  "<mask_269>": 50931,
+  "<mask_26>": 51174,
+  "<mask_270>": 50930,
+  "<mask_271>": 50929,
+  "<mask_272>": 50928,
+  "<mask_273>": 50927,
+  "<mask_274>": 50926,
+  "<mask_275>": 50925,
+  "<mask_276>": 50924,
+  "<mask_277>": 50923,
+  "<mask_278>": 50922,
+  "<mask_279>": 50921,
+  "<mask_27>": 51173,
+  "<mask_280>": 50920,
+  "<mask_281>": 50919,
+  "<mask_282>": 50918,
+  "<mask_283>": 50917,
+  "<mask_284>": 50916,
+  "<mask_285>": 50915,
+  "<mask_286>": 50914,
+  "<mask_287>": 50913,
+  "<mask_288>": 50912,
+  "<mask_289>": 50911,
+  "<mask_28>": 51172,
+  "<mask_290>": 50910,
+  "<mask_291>": 50909,
+  "<mask_292>": 50908,
+  "<mask_293>": 50907,
+  "<mask_294>": 50906,
+  "<mask_295>": 50905,
+  "<mask_296>": 50904,
+  "<mask_297>": 50903,
+  "<mask_298>": 50902,
+  "<mask_299>": 50901,
+  "<mask_29>": 51171,
+  "<mask_2>": 51198,
+  "<mask_300>": 50900,
+  "<mask_301>": 50899,
+  "<mask_302>": 50898,
+  "<mask_303>": 50897,
+  "<mask_304>": 50896,
+  "<mask_305>": 50895,
+  "<mask_306>": 50894,
+  "<mask_307>": 50893,
+  "<mask_308>": 50892,
+  "<mask_309>": 50891,
+  "<mask_30>": 51170,
+  "<mask_310>": 50890,
+  "<mask_311>": 50889,
+  "<mask_312>": 50888,
+  "<mask_313>": 50887,
+  "<mask_314>": 50886,
+  "<mask_315>": 50885,
+  "<mask_316>": 50884,
+  "<mask_317>": 50883,
+  "<mask_318>": 50882,
+  "<mask_319>": 50881,
+  "<mask_31>": 51169,
+  "<mask_320>": 50880,
+  "<mask_321>": 50879,
+  "<mask_322>": 50878,
+  "<mask_323>": 50877,
+  "<mask_324>": 50876,
+  "<mask_325>": 50875,
+  "<mask_326>": 50874,
+  "<mask_327>": 50873,
+  "<mask_328>": 50872,
+  "<mask_329>": 50871,
+  "<mask_32>": 51168,
+  "<mask_330>": 50870,
+  "<mask_331>": 50869,
+  "<mask_332>": 50868,
+  "<mask_333>": 50867,
+  "<mask_334>": 50866,
+  "<mask_335>": 50865,
+  "<mask_336>": 50864,
+  "<mask_337>": 50863,
+  "<mask_338>": 50862,
+  "<mask_339>": 50861,
+  "<mask_33>": 51167,
+  "<mask_340>": 50860,
+  "<mask_341>": 50859,
+  "<mask_342>": 50858,
+  "<mask_343>": 50857,
+  "<mask_344>": 50856,
+  "<mask_345>": 50855,
+  "<mask_346>": 50854,
+  "<mask_347>": 50853,
+  "<mask_348>": 50852,
+  "<mask_349>": 50851,
+  "<mask_34>": 51166,
+  "<mask_350>": 50850,
+  "<mask_351>": 50849,
+  "<mask_352>": 50848,
+  "<mask_353>": 50847,
+  "<mask_354>": 50846,
+  "<mask_355>": 50845,
+  "<mask_356>": 50844,
+  "<mask_357>": 50843,
+  "<mask_358>": 50842,
+  "<mask_359>": 50841,
+  "<mask_35>": 51165,
+  "<mask_360>": 50840,
+  "<mask_361>": 50839,
+  "<mask_362>": 50838,
+  "<mask_363>": 50837,
+  "<mask_364>": 50836,
+  "<mask_365>": 50835,
+  "<mask_366>": 50834,
+  "<mask_367>": 50833,
+  "<mask_368>": 50832,
+  "<mask_369>": 50831,
+  "<mask_36>": 51164,
+  "<mask_370>": 50830,
+  "<mask_371>": 50829,
+  "<mask_372>": 50828,
+  "<mask_373>": 50827,
+  "<mask_374>": 50826,
+  "<mask_375>": 50825,
+  "<mask_376>": 50824,
+  "<mask_377>": 50823,
+  "<mask_378>": 50822,
+  "<mask_379>": 50821,
+  "<mask_37>": 51163,
+  "<mask_380>": 50820,
+  "<mask_381>": 50819,
+  "<mask_382>": 50818,
+  "<mask_383>": 50817,
+  "<mask_384>": 50816,
+  "<mask_385>": 50815,
+  "<mask_386>": 50814,
+  "<mask_387>": 50813,
+  "<mask_388>": 50812,
+  "<mask_389>": 50811,
+  "<mask_38>": 51162,
+  "<mask_390>": 50810,
+  "<mask_391>": 50809,
+  "<mask_392>": 50808,
+  "<mask_393>": 50807,
+  "<mask_394>": 50806,
+  "<mask_395>": 50805,
+  "<mask_396>": 50804,
+  "<mask_397>": 50803,
+  "<mask_398>": 50802,
+  "<mask_399>": 50801,
+  "<mask_39>": 51161,
+  "<mask_3>": 51197,
+  "<mask_400>": 50800,
+  "<mask_401>": 50799,
+  "<mask_402>": 50798,
+  "<mask_403>": 50797,
+  "<mask_404>": 50796,
+  "<mask_405>": 50795,
+  "<mask_406>": 50794,
+  "<mask_407>": 50793,
+  "<mask_408>": 50792,
+  "<mask_409>": 50791,
+  "<mask_40>": 51160,
+  "<mask_410>": 50790,
+  "<mask_411>": 50789,
+  "<mask_412>": 50788,
+  "<mask_413>": 50787,
+  "<mask_414>": 50786,
+  "<mask_415>": 50785,
+  "<mask_416>": 50784,
+  "<mask_417>": 50783,
+  "<mask_418>": 50782,
+  "<mask_419>": 50781,
+  "<mask_41>": 51159,
+  "<mask_420>": 50780,
+  "<mask_421>": 50779,
+  "<mask_422>": 50778,
+  "<mask_423>": 50777,
+  "<mask_424>": 50776,
+  "<mask_425>": 50775,
+  "<mask_426>": 50774,
+  "<mask_427>": 50773,
+  "<mask_428>": 50772,
+  "<mask_429>": 50771,
+  "<mask_42>": 51158,
+  "<mask_430>": 50770,
+  "<mask_431>": 50769,
+  "<mask_432>": 50768,
+  "<mask_433>": 50767,
+  "<mask_434>": 50766,
+  "<mask_435>": 50765,
+  "<mask_436>": 50764,
+  "<mask_437>": 50763,
+  "<mask_438>": 50762,
+  "<mask_439>": 50761,
+  "<mask_43>": 51157,
+  "<mask_440>": 50760,
+  "<mask_441>": 50759,
+  "<mask_442>": 50758,
+  "<mask_443>": 50757,
+  "<mask_444>": 50756,
+  "<mask_445>": 50755,
+  "<mask_446>": 50754,
+  "<mask_447>": 50753,
+  "<mask_448>": 50752,
+  "<mask_449>": 50751,
+  "<mask_44>": 51156,
+  "<mask_450>": 50750,
+  "<mask_451>": 50749,
+  "<mask_452>": 50748,
+  "<mask_453>": 50747,
+  "<mask_454>": 50746,
+  "<mask_455>": 50745,
+  "<mask_456>": 50744,
+  "<mask_457>": 50743,
+  "<mask_458>": 50742,
+  "<mask_459>": 50741,
+  "<mask_45>": 51155,
+  "<mask_460>": 50740,
+  "<mask_461>": 50739,
+  "<mask_462>": 50738,
+  "<mask_463>": 50737,
+  "<mask_464>": 50736,
+  "<mask_465>": 50735,
+  "<mask_466>": 50734,
+  "<mask_467>": 50733,
+  "<mask_468>": 50732,
+  "<mask_469>": 50731,
+  "<mask_46>": 51154,
+  "<mask_470>": 50730,
+  "<mask_471>": 50729,
+  "<mask_472>": 50728,
+  "<mask_473>": 50727,
+  "<mask_474>": 50726,
+  "<mask_475>": 50725,
+  "<mask_476>": 50724,
+  "<mask_477>": 50723,
+  "<mask_478>": 50722,
+  "<mask_479>": 50721,
+  "<mask_47>": 51153,
+  "<mask_480>": 50720,
+  "<mask_481>": 50719,
+  "<mask_482>": 50718,
+  "<mask_483>": 50717,
+  "<mask_484>": 50716,
+  "<mask_485>": 50715,
+  "<mask_486>": 50714,
+  "<mask_487>": 50713,
+  "<mask_488>": 50712,
+  "<mask_489>": 50711,
+  "<mask_48>": 51152,
+  "<mask_490>": 50710,
+  "<mask_491>": 50709,
+  "<mask_492>": 50708,
+  "<mask_493>": 50707,
+  "<mask_494>": 50706,
+  "<mask_495>": 50705,
+  "<mask_496>": 50704,
+  "<mask_497>": 50703,
+  "<mask_498>": 50702,
+  "<mask_499>": 50701,
+  "<mask_49>": 51151,
+  "<mask_4>": 51196,
+  "<mask_500>": 50700,
+  "<mask_501>": 50699,
+  "<mask_502>": 50698,
+  "<mask_503>": 50697,
+  "<mask_504>": 50696,
+  "<mask_505>": 50695,
+  "<mask_506>": 50694,
+  "<mask_507>": 50693,
+  "<mask_508>": 50692,
+  "<mask_509>": 50691,
+  "<mask_50>": 51150,
+  "<mask_510>": 50690,
+  "<mask_511>": 50689,
+  "<mask_512>": 50688,
+  "<mask_513>": 50687,
+  "<mask_514>": 50686,
+  "<mask_515>": 50685,
+  "<mask_516>": 50684,
+  "<mask_517>": 50683,
+  "<mask_518>": 50682,
+  "<mask_519>": 50681,
+  "<mask_51>": 51149,
+  "<mask_520>": 50680,
+  "<mask_521>": 50679,
+  "<mask_522>": 50678,
+  "<mask_523>": 50677,
+  "<mask_524>": 50676,
+  "<mask_525>": 50675,
+  "<mask_526>": 50674,
+  "<mask_527>": 50673,
+  "<mask_528>": 50672,
+  "<mask_529>": 50671,
+  "<mask_52>": 51148,
+  "<mask_530>": 50670,
+  "<mask_531>": 50669,
+  "<mask_532>": 50668,
+  "<mask_533>": 50667,
+  "<mask_534>": 50666,
+  "<mask_535>": 50665,
+  "<mask_536>": 50664,
+  "<mask_537>": 50663,
+  "<mask_538>": 50662,
+  "<mask_539>": 50661,
+  "<mask_53>": 51147,
+  "<mask_540>": 50660,
+  "<mask_541>": 50659,
+  "<mask_542>": 50658,
+  "<mask_543>": 50657,
+  "<mask_544>": 50656,
+  "<mask_545>": 50655,
+  "<mask_546>": 50654,
+  "<mask_547>": 50653,
+  "<mask_548>": 50652,
+  "<mask_549>": 50651,
+  "<mask_54>": 51146,
+  "<mask_550>": 50650,
+  "<mask_551>": 50649,
+  "<mask_552>": 50648,
+  "<mask_553>": 50647,
+  "<mask_554>": 50646,
+  "<mask_555>": 50645,
+  "<mask_556>": 50644,
+  "<mask_557>": 50643,
+  "<mask_558>": 50642,
+  "<mask_559>": 50641,
+  "<mask_55>": 51145,
+  "<mask_560>": 50640,
+  "<mask_561>": 50639,
+  "<mask_562>": 50638,
+  "<mask_563>": 50637,
+  "<mask_564>": 50636,
+  "<mask_565>": 50635,
+  "<mask_566>": 50634,
+  "<mask_567>": 50633,
+  "<mask_568>": 50632,
+  "<mask_569>": 50631,
+  "<mask_56>": 51144,
+  "<mask_570>": 50630,
+  "<mask_571>": 50629,
+  "<mask_572>": 50628,
+  "<mask_573>": 50627,
+  "<mask_574>": 50626,
+  "<mask_575>": 50625,
+  "<mask_576>": 50624,
+  "<mask_577>": 50623,
+  "<mask_578>": 50622,
+  "<mask_579>": 50621,
+  "<mask_57>": 51143,
+  "<mask_580>": 50620,
+  "<mask_581>": 50619,
+  "<mask_582>": 50618,
+  "<mask_583>": 50617,
+  "<mask_584>": 50616,
+  "<mask_585>": 50615,
+  "<mask_586>": 50614,
+  "<mask_587>": 50613,
+  "<mask_588>": 50612,
+  "<mask_589>": 50611,
+  "<mask_58>": 51142,
+  "<mask_590>": 50610,
+  "<mask_591>": 50609,
+  "<mask_592>": 50608,
+  "<mask_593>": 50607,
+  "<mask_594>": 50606,
+  "<mask_595>": 50605,
+  "<mask_596>": 50604,
+  "<mask_597>": 50603,
+  "<mask_598>": 50602,
+  "<mask_599>": 50601,
+  "<mask_59>": 51141,
+  "<mask_5>": 51195,
+  "<mask_600>": 50600,
+  "<mask_601>": 50599,
+  "<mask_602>": 50598,
+  "<mask_603>": 50597,
+  "<mask_604>": 50596,
+  "<mask_605>": 50595,
+  "<mask_606>": 50594,
+  "<mask_607>": 50593,
+  "<mask_608>": 50592,
+  "<mask_609>": 50591,
+  "<mask_60>": 51140,
+  "<mask_610>": 50590,
+  "<mask_611>": 50589,
+  "<mask_612>": 50588,
+  "<mask_613>": 50587,
+  "<mask_614>": 50586,
+  "<mask_615>": 50585,
+  "<mask_616>": 50584,
+  "<mask_617>": 50583,
+  "<mask_618>": 50582,
+  "<mask_619>": 50581,
+  "<mask_61>": 51139,
+  "<mask_620>": 50580,
+  "<mask_621>": 50579,
+  "<mask_622>": 50578,
+  "<mask_623>": 50577,
+  "<mask_624>": 50576,
+  "<mask_625>": 50575,
+  "<mask_626>": 50574,
+  "<mask_627>": 50573,
+  "<mask_628>": 50572,
+  "<mask_629>": 50571,
+  "<mask_62>": 51138,
+  "<mask_630>": 50570,
+  "<mask_631>": 50569,
+  "<mask_632>": 50568,
+  "<mask_633>": 50567,
+  "<mask_634>": 50566,
+  "<mask_635>": 50565,
+  "<mask_636>": 50564,
+  "<mask_637>": 50563,
+  "<mask_638>": 50562,
+  "<mask_639>": 50561,
+  "<mask_63>": 51137,
+  "<mask_640>": 50560,
+  "<mask_641>": 50559,
+  "<mask_642>": 50558,
+  "<mask_643>": 50557,
+  "<mask_644>": 50556,
+  "<mask_645>": 50555,
+  "<mask_646>": 50554,
+  "<mask_647>": 50553,
+  "<mask_648>": 50552,
+  "<mask_649>": 50551,
+  "<mask_64>": 51136,
+  "<mask_650>": 50550,
+  "<mask_651>": 50549,
+  "<mask_652>": 50548,
+  "<mask_653>": 50547,
+  "<mask_654>": 50546,
+  "<mask_655>": 50545,
+  "<mask_656>": 50544,
+  "<mask_657>": 50543,
+  "<mask_658>": 50542,
+  "<mask_659>": 50541,
+  "<mask_65>": 51135,
+  "<mask_660>": 50540,
+  "<mask_661>": 50539,
+  "<mask_662>": 50538,
+  "<mask_663>": 50537,
+  "<mask_664>": 50536,
+  "<mask_665>": 50535,
+  "<mask_666>": 50534,
+  "<mask_667>": 50533,
+  "<mask_668>": 50532,
+  "<mask_669>": 50531,
+  "<mask_66>": 51134,
+  "<mask_670>": 50530,
+  "<mask_671>": 50529,
+  "<mask_672>": 50528,
+  "<mask_673>": 50527,
+  "<mask_674>": 50526,
+  "<mask_675>": 50525,
+  "<mask_676>": 50524,
+  "<mask_677>": 50523,
+  "<mask_678>": 50522,
+  "<mask_679>": 50521,
+  "<mask_67>": 51133,
+  "<mask_680>": 50520,
+  "<mask_681>": 50519,
+  "<mask_682>": 50518,
+  "<mask_683>": 50517,
+  "<mask_684>": 50516,
+  "<mask_685>": 50515,
+  "<mask_686>": 50514,
+  "<mask_687>": 50513,
+  "<mask_688>": 50512,
+  "<mask_689>": 50511,
+  "<mask_68>": 51132,
+  "<mask_690>": 50510,
+  "<mask_691>": 50509,
+  "<mask_692>": 50508,
+  "<mask_693>": 50507,
+  "<mask_694>": 50506,
+  "<mask_695>": 50505,
+  "<mask_696>": 50504,
+  "<mask_697>": 50503,
+  "<mask_698>": 50502,
+  "<mask_699>": 50501,
+  "<mask_69>": 51131,
+  "<mask_6>": 51194,
+  "<mask_700>": 50500,
+  "<mask_701>": 50499,
+  "<mask_702>": 50498,
+  "<mask_703>": 50497,
+  "<mask_704>": 50496,
+  "<mask_705>": 50495,
+  "<mask_706>": 50494,
+  "<mask_707>": 50493,
+  "<mask_708>": 50492,
+  "<mask_709>": 50491,
+  "<mask_70>": 51130,
+  "<mask_710>": 50490,
+  "<mask_711>": 50489,
+  "<mask_712>": 50488,
+  "<mask_713>": 50487,
+  "<mask_714>": 50486,
+  "<mask_715>": 50485,
+  "<mask_716>": 50484,
+  "<mask_717>": 50483,
+  "<mask_718>": 50482,
+  "<mask_719>": 50481,
+  "<mask_71>": 51129,
+  "<mask_720>": 50480,
+  "<mask_721>": 50479,
+  "<mask_722>": 50478,
+  "<mask_723>": 50477,
+  "<mask_724>": 50476,
+  "<mask_725>": 50475,
+  "<mask_726>": 50474,
+  "<mask_727>": 50473,
+  "<mask_728>": 50472,
+  "<mask_729>": 50471,
+  "<mask_72>": 51128,
+  "<mask_730>": 50470,
+  "<mask_731>": 50469,
+  "<mask_732>": 50468,
+  "<mask_733>": 50467,
+  "<mask_734>": 50466,
+  "<mask_735>": 50465,
+  "<mask_736>": 50464,
+  "<mask_737>": 50463,
+  "<mask_738>": 50462,
+  "<mask_739>": 50461,
+  "<mask_73>": 51127,
+  "<mask_740>": 50460,
+  "<mask_741>": 50459,
+  "<mask_742>": 50458,
+  "<mask_743>": 50457,
+  "<mask_744>": 50456,
+  "<mask_745>": 50455,
+  "<mask_746>": 50454,
+  "<mask_747>": 50453,
+  "<mask_748>": 50452,
+  "<mask_749>": 50451,
+  "<mask_74>": 51126,
+  "<mask_750>": 50450,
+  "<mask_751>": 50449,
+  "<mask_752>": 50448,
+  "<mask_753>": 50447,
+  "<mask_754>": 50446,
+  "<mask_755>": 50445,
+  "<mask_756>": 50444,
+  "<mask_757>": 50443,
+  "<mask_758>": 50442,
+  "<mask_759>": 50441,
+  "<mask_75>": 51125,
+  "<mask_760>": 50440,
+  "<mask_761>": 50439,
+  "<mask_762>": 50438,
+  "<mask_763>": 50437,
+  "<mask_764>": 50436,
+  "<mask_765>": 50435,
+  "<mask_766>": 50434,
+  "<mask_767>": 50433,
+  "<mask_768>": 50432,
+  "<mask_769>": 50431,
+  "<mask_76>": 51124,
+  "<mask_770>": 50430,
+  "<mask_771>": 50429,
+  "<mask_772>": 50428,
+  "<mask_773>": 50427,
+  "<mask_774>": 50426,
+  "<mask_775>": 50425,
+  "<mask_776>": 50424,
+  "<mask_777>": 50423,
+  "<mask_778>": 50422,
+  "<mask_779>": 50421,
+  "<mask_77>": 51123,
+  "<mask_780>": 50420,
+  "<mask_781>": 50419,
+  "<mask_782>": 50418,
+  "<mask_783>": 50417,
+  "<mask_784>": 50416,
+  "<mask_785>": 50415,
+  "<mask_786>": 50414,
+  "<mask_787>": 50413,
+  "<mask_788>": 50412,
+  "<mask_789>": 50411,
+  "<mask_78>": 51122,
+  "<mask_790>": 50410,
+  "<mask_791>": 50409,
+  "<mask_792>": 50408,
+  "<mask_793>": 50407,
+  "<mask_794>": 50406,
+  "<mask_795>": 50405,
+  "<mask_796>": 50404,
+  "<mask_797>": 50403,
+  "<mask_798>": 50402,
+  "<mask_799>": 50401,
+  "<mask_79>": 51121,
+  "<mask_7>": 51193,
+  "<mask_800>": 50400,
+  "<mask_801>": 50399,
+  "<mask_802>": 50398,
+  "<mask_803>": 50397,
+  "<mask_804>": 50396,
+  "<mask_805>": 50395,
+  "<mask_806>": 50394,
+  "<mask_807>": 50393,
+  "<mask_808>": 50392,
+  "<mask_809>": 50391,
+  "<mask_80>": 51120,
+  "<mask_810>": 50390,
+  "<mask_811>": 50389,
+  "<mask_812>": 50388,
+  "<mask_813>": 50387,
+  "<mask_814>": 50386,
+  "<mask_815>": 50385,
+  "<mask_816>": 50384,
+  "<mask_817>": 50383,
+  "<mask_818>": 50382,
+  "<mask_819>": 50381,
+  "<mask_81>": 51119,
+  "<mask_820>": 50380,
+  "<mask_821>": 50379,
+  "<mask_822>": 50378,
+  "<mask_823>": 50377,
+  "<mask_824>": 50376,
+  "<mask_825>": 50375,
+  "<mask_826>": 50374,
+  "<mask_827>": 50373,
+  "<mask_828>": 50372,
+  "<mask_829>": 50371,
+  "<mask_82>": 51118,
+  "<mask_830>": 50370,
+  "<mask_831>": 50369,
+  "<mask_832>": 50368,
+  "<mask_833>": 50367,
+  "<mask_834>": 50366,
+  "<mask_835>": 50365,
+  "<mask_836>": 50364,
+  "<mask_837>": 50363,
+  "<mask_838>": 50362,
+  "<mask_839>": 50361,
+  "<mask_83>": 51117,
+  "<mask_840>": 50360,
+  "<mask_841>": 50359,
+  "<mask_842>": 50358,
+  "<mask_843>": 50357,
+  "<mask_844>": 50356,
+  "<mask_845>": 50355,
+  "<mask_846>": 50354,
+  "<mask_847>": 50353,
+  "<mask_848>": 50352,
+  "<mask_849>": 50351,
+  "<mask_84>": 51116,
+  "<mask_850>": 50350,
+  "<mask_851>": 50349,
+  "<mask_852>": 50348,
+  "<mask_853>": 50347,
+  "<mask_854>": 50346,
+  "<mask_855>": 50345,
+  "<mask_856>": 50344,
+  "<mask_857>": 50343,
+  "<mask_858>": 50342,
+  "<mask_859>": 50341,
+  "<mask_85>": 51115,
+  "<mask_860>": 50340,
+  "<mask_861>": 50339,
+  "<mask_862>": 50338,
+  "<mask_863>": 50337,
+  "<mask_864>": 50336,
+  "<mask_865>": 50335,
+  "<mask_866>": 50334,
+  "<mask_867>": 50333,
+  "<mask_868>": 50332,
+  "<mask_869>": 50331,
+  "<mask_86>": 51114,
+  "<mask_870>": 50330,
+  "<mask_871>": 50329,
+  "<mask_872>": 50328,
+  "<mask_873>": 50327,
+  "<mask_874>": 50326,
+  "<mask_875>": 50325,
+  "<mask_876>": 50324,
+  "<mask_877>": 50323,
+  "<mask_878>": 50322,
+  "<mask_879>": 50321,
+  "<mask_87>": 51113,
+  "<mask_880>": 50320,
+  "<mask_881>": 50319,
+  "<mask_882>": 50318,
+  "<mask_883>": 50317,
+  "<mask_884>": 50316,
+  "<mask_885>": 50315,
+  "<mask_886>": 50314,
+  "<mask_887>": 50313,
+  "<mask_888>": 50312,
+  "<mask_889>": 50311,
+  "<mask_88>": 51112,
+  "<mask_890>": 50310,
+  "<mask_891>": 50309,
+  "<mask_892>": 50308,
+  "<mask_893>": 50307,
+  "<mask_894>": 50306,
+  "<mask_895>": 50305,
+  "<mask_896>": 50304,
+  "<mask_897>": 50303,
+  "<mask_898>": 50302,
+  "<mask_899>": 50301,
+  "<mask_89>": 51111,
+  "<mask_8>": 51192,
+  "<mask_90>": 51110,
+  "<mask_91>": 51109,
+  "<mask_92>": 51108,
+  "<mask_93>": 51107,
+  "<mask_94>": 51106,
+  "<mask_95>": 51105,
+  "<mask_96>": 51104,
+  "<mask_97>": 51103,
+  "<mask_98>": 51102,
+  "<mask_99>": 51101,
+  "<mask_9>": 51191,
+  "<sep>": 50299
+}

config.json ADDED Viewed

	@@ -0,0 +1,48 @@

+{
+  "_name_or_path": "checkpoints/codegen2-1B/",
+  "activation_function": "gelu_new",
+  "architectures": [
+    "CodeGenForCausalLM"
+  ],
+  "attn_pdrop": 0.0,
+  "auto_map": {
+    "AutoConfig": "configuration_codegen.CodeGenConfig",
+    "AutoModel": "modeling_codegen.CodeGenModel",
+    "AutoModelForCausalLM": "modeling_codegen.CodeGenForCausalLM"
+  },
+  "bos_token_id": 1,
+  "embd_pdrop": 0.0,
+  "eos_token_id": 2,
+  "gradient_checkpointing": false,
+  "head_dim": 128,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "codegen",
+  "n_ctx": 2048,
+  "n_embd": 2048,
+  "n_head": 16,
+  "n_inner": null,
+  "n_layer": 16,
+  "n_positions": 2048,
+  "resid_pdrop": 0.0,
+  "rotary_dim": 64,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "task_specific_params": {
+    "text-generation": {
+      "do_sample": true,
+      "max_length": 50,
+      "temperature": 1.0
+    }
+  },
+  "tie_word_embeddings": false,
+  "tokenizer_class": "GPT2Tokenizer",
+  "torch_dtype": "float32",
+  "transformers_version": "4.25.1",
+  "use_cache": true,
+  "vocab_size": 51200
+}

configuration_codegen.py ADDED Viewed

	@@ -0,0 +1,220 @@

+# coding=utf-8
+# Copyright 2022 Salesforce authors, The EleutherAI, and HuggingFace Teams. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" CodeGen model configuration"""
+from collections import OrderedDict
+from typing import Any, List, Mapping, Optional
+from transformers import PreTrainedTokenizer, TensorType, is_torch_available
+from transformers.configuration_utils import PretrainedConfig
+from transformers.onnx import OnnxConfigWithPast, PatchingSpec
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+class CodeGenConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`CodeGenModel`]. It is used to instantiate a
+    CodeGen model according to the specified arguments, defining the model architecture. Instantiating a configuration
+    with the defaults will yield a similar configuration to that of the CodeGen
+    [Salesforce/codegen-2B-mono](https://huggingface.co/Salesforce/codegen-2B-mono) architecture. Configuration objects
+    inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the documentation from
+    [`PretrainedConfig`] for more information.
+    Args:
+        vocab_size (`int`, *optional*, defaults to 50400):
+            Vocabulary size of the CodeGen model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`CodeGenModel`].
+        n_positions (`int`, *optional*, defaults to 2048):
+            The maximum sequence length that this model might ever be used with. Typically set this to something large
+            just in case (e.g., 512 or 1024 or 2048).
+        n_embd (`int`, *optional*, defaults to 4096):
+            Dimensionality of the embeddings and hidden states.
+        n_layer (`int`, *optional*, defaults to 28):
+            Number of hidden layers in the Transformer encoder.
+        n_head (`int`, *optional*, defaults to 16):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        rotary_dim (`int`, *optional*, defaults to 64):
+            Number of dimensions in the embedding that Rotary Position Embedding is applied to.
+        n_inner (`int`, *optional*, defaults to None):
+            Dimensionality of the inner feed-forward layers. `None` will set it to 4 times n_embd
+        activation_function (`str`, *optional*, defaults to `"gelu_new"`):
+            Activation function, to be selected in the list `["relu", "silu", "gelu", "tanh", "gelu_new"]`.
+        resid_pdrop (`float`, *optional*, defaults to 0.1):
+            The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
+        embd_pdrop (`int`, *optional*, defaults to 0.1):
+            The dropout ratio for the embeddings.
+        attn_pdrop (`float`, *optional*, defaults to 0.1):
+            The dropout ratio for the attention.
+        layer_norm_epsilon (`float`, *optional*, defaults to 1e-5):
+            The epsilon to use in the layer normalization layers.
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        scale_attn_weights (`bool`, *optional*, defaults to `True`):
+            Scale attention weights by dividing by sqrt(hidden_size).
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether or not the model should return the last key/values attentions (not used by all models).
+    Example:
+    ```python
+    >>> from transformers import CodeGenModel, CodeGenConfig
+    >>> # Initializing a CodeGen 6B configuration
+    >>> configuration = CodeGenConfig()
+    >>> # Initializing a model from the configuration
+    >>> model = CodeGenModel(configuration)
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+    model_type = "codegen"
+    attribute_map = {
+        "max_position_embeddings": "n_positions",
+        "hidden_size": "n_embd",
+        "num_attention_heads": "n_head",
+        "num_hidden_layers": "n_layer",
+    }
+    def __init__(
+        self,
+        vocab_size=50400,
+        n_positions=2048,
+        n_ctx=2048,
+        n_embd=4096,
+        n_layer=28,
+        n_head=16,
+        rotary_dim=64,
+        n_inner=None,
+        activation_function="gelu_new",
+        resid_pdrop=0.0,
+        embd_pdrop=0.0,
+        attn_pdrop=0.0,
+        layer_norm_epsilon=1e-5,
+        initializer_range=0.02,
+        scale_attn_weights=True,
+        use_cache=True,
+        bos_token_id=50256,
+        eos_token_id=50256,
+        tie_word_embeddings=False,
+        **kwargs
+    ):
+        self.vocab_size = vocab_size
+        self.n_ctx = n_ctx
+        self.n_positions = n_positions
+        self.n_embd = n_embd
+        self.n_layer = n_layer
+        self.n_head = n_head
+        self.n_inner = n_inner
+        self.rotary_dim = rotary_dim
+        self.activation_function = activation_function
+        self.resid_pdrop = resid_pdrop
+        self.embd_pdrop = embd_pdrop
+        self.attn_pdrop = attn_pdrop
+        self.layer_norm_epsilon = layer_norm_epsilon
+        self.initializer_range = initializer_range
+        self.scale_attn_weights = scale_attn_weights
+        self.use_cache = use_cache
+        self.bos_token_id = bos_token_id
+        self.eos_token_id = eos_token_id
+        super().__init__(
+            bos_token_id=bos_token_id, eos_token_id=eos_token_id, tie_word_embeddings=tie_word_embeddings, **kwargs
+        )
+# Copied from transformers.models.gpt2.configuration_gpt2.GPT2OnnxConfig
+class CodeGenOnnxConfig(OnnxConfigWithPast):
+    def __init__(
+        self,
+        config: PretrainedConfig,
+        task: str = "default",
+        patching_specs: List[PatchingSpec] = None,
+        use_past: bool = False,
+    ):
+        super().__init__(config, task=task, patching_specs=patching_specs, use_past=use_past)
+        if not getattr(self._config, "pad_token_id", None):
+            # TODO: how to do that better?
+            self._config.pad_token_id = 0
+    @property
+    def inputs(self) -> Mapping[str, Mapping[int, str]]:
+        common_inputs = OrderedDict({"input_ids": {0: "batch", 1: "sequence"}})
+        if self.use_past:
+            self.fill_with_past_key_values_(common_inputs, direction="inputs")
+            common_inputs["attention_mask"] = {0: "batch", 1: "past_sequence + sequence"}
+        else:
+            common_inputs["attention_mask"] = {0: "batch", 1: "sequence"}
+        return common_inputs
+    @property
+    def num_layers(self) -> int:
+        return self._config.n_layer
+    @property
+    def num_attention_heads(self) -> int:
+        return self._config.n_head
+    def generate_dummy_inputs(
+        self,
+        tokenizer: PreTrainedTokenizer,
+        batch_size: int = -1,
+        seq_length: int = -1,
+        is_pair: bool = False,
+        framework: Optional[TensorType] = None,
+    ) -> Mapping[str, Any]:
+        common_inputs = super(OnnxConfigWithPast, self).generate_dummy_inputs(
+            tokenizer, batch_size=batch_size, seq_length=seq_length, is_pair=is_pair, framework=framework
+        )
+        # We need to order the input in the way they appears in the forward()
+        ordered_inputs = OrderedDict({"input_ids": common_inputs["input_ids"]})
+        # Need to add the past_keys
+        if self.use_past:
+            if not is_torch_available():
+                raise ValueError("Cannot generate dummy past_keys inputs without PyTorch installed.")
+            else:
+                import torch
+                batch, seqlen = common_inputs["input_ids"].shape
+                # Not using the same length for past_key_values
+                past_key_values_length = seqlen + 2
+                past_shape = (
+                    batch,
+                    self.num_attention_heads,
+                    past_key_values_length,
+                    self._config.hidden_size // self.num_attention_heads,
+                )
+                ordered_inputs["past_key_values"] = [
+                    (torch.zeros(past_shape), torch.zeros(past_shape)) for _ in range(self.num_layers)
+                ]
+        ordered_inputs["attention_mask"] = common_inputs["attention_mask"]
+        if self.use_past:
+            mask_dtype = ordered_inputs["attention_mask"].dtype
+            ordered_inputs["attention_mask"] = torch.cat(
+                [ordered_inputs["attention_mask"], torch.ones(batch, past_key_values_length, dtype=mask_dtype)], dim=1
+            )
+        return ordered_inputs
+    @property
+    def default_onnx_opset(self) -> int:
+        return 13

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

modeling_codegen.py ADDED Viewed

	@@ -0,0 +1,747 @@

+# coding=utf-8
+# Copyright 2022 Salesforce authors, The EleutherAI, and HuggingFace Teams. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" PyTorch CodeGen model."""
+from typing import Optional, Tuple, Union
+import torch
+import torch.utils.checkpoint
+from torch import nn
+from torch.nn import CrossEntropyLoss
+from transformers.activations import ACT2FN
+from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast
+from transformers.modeling_utils import PreTrainedModel
+from transformers.utils import add_code_sample_docstrings, add_start_docstrings, add_start_docstrings_to_model_forward, logging
+from .configuration_codegen import CodeGenConfig
+logger = logging.get_logger(__name__)
+_CHECKPOINT_FOR_DOC = "Salesforce/codegen-2B-mono"
+_CONFIG_FOR_DOC = "CodeGenConfig"
+_TOKENIZER_FOR_DOC = "GPT2Tokenizer"
+CODEGEN_PRETRAINED_MODEL_ARCHIVE_LIST = [
+    "Salesforce/codegen-350M-nl",
+    "Salesforce/codegen-350M-multi",
+    "Salesforce/codegen-350M-mono",
+    "Salesforce/codegen-2B-nl",
+    "Salesforce/codegen-2B-multi",
+    "Salesforce/codegen-2B-mono",
+    "Salesforce/codegen-6B-nl",
+    "Salesforce/codegen-6B-multi",
+    "Salesforce/codegen-6B-mono",
+    "Salesforce/codegen-16B-nl",
+    "Salesforce/codegen-16B-multi",
+    "Salesforce/codegen-16B-mono",
+    # See all CodeGen models at https://huggingface.co/models?filter=codegen
+]
+# Copied from transformers.models.gptj.modeling_gptj.fixed_pos_embedding
+def fixed_pos_embedding(x, seq_dim=1, seq_len=None):
+    dim = x.shape[-1]
+    if seq_len is None:
+        seq_len = x.shape[seq_dim]
+    inv_freq = 1.0 / (10000 ** (torch.arange(0, dim, 2) / dim))
+    sinusoid_inp = (
+        torch.einsum("i , j -> i j", torch.arange(seq_len, dtype=torch.float), inv_freq).to(x.device).float()
+    )
+    return torch.sin(sinusoid_inp), torch.cos(sinusoid_inp)
+# Copied from transformers.models.gptj.modeling_gptj.rotate_every_two
+def rotate_every_two(x):
+    x1 = x[:, :, :, ::2]
+    x2 = x[:, :, :, 1::2]
+    x = torch.stack((-x2, x1), dim=-1)
+    return x.flatten(-2)  # in einsum notation: rearrange(x, '... d j -> ... (d j)')
+# Copied from transformers.models.gptj.modeling_gptj.duplicate_interleave
+def duplicate_interleave(m):
+    """
+    A simple version of `torch.repeat_interleave` for duplicating a matrix while interleaving the copy.
+    """
+    dim0 = m.shape[0]
+    m = m.view(-1, 1)  # flatten the matrix
+    m = m.repeat(1, 2)  # repeat all elements into the 2nd dimension
+    m = m.view(dim0, -1)  # reshape into a matrix, interleaving the copy
+    return m
+# Copied from transformers.models.gptj.modeling_gptj.apply_rotary_pos_emb
+def apply_rotary_pos_emb(x, sincos, offset=0):
+    sin, cos = map(lambda t: duplicate_interleave(t)[None, offset : x.shape[1] + offset, None, :], sincos)
+    # einsum notation for lambda t: repeat(t[offset:x.shape[1]+offset,:], "n d -> () n () (d j)", j=2)
+    return (x * cos) + (rotate_every_two(x) * sin)
+class CodeGenAttention(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        max_positions = config.max_position_embeddings
+        self.register_buffer(
+            "causal_mask",
+            torch.tril(torch.ones((max_positions, max_positions), dtype=torch.bool)).view(
+                1, 1, max_positions, max_positions
+            ),
+        )
+        self.attn_dropout = nn.Dropout(config.attn_pdrop)
+        self.resid_dropout = nn.Dropout(config.resid_pdrop)
+        self.embed_dim = config.hidden_size
+        self.num_attention_heads = config.num_attention_heads
+        self.head_dim = self.embed_dim // self.num_attention_heads
+        if self.head_dim * self.num_attention_heads != self.embed_dim:
+            raise ValueError(
+                f"embed_dim must be divisible by num_attention_heads (got `embed_dim`: {self.embed_dim} and"
+                f" `num_attention_heads`: {self.num_attention_heads})."
+            )
+        self.scale_attn = torch.sqrt(torch.tensor(self.head_dim, dtype=torch.float32)).to(torch.get_default_dtype())
+        self.qkv_proj = nn.Linear(self.embed_dim, self.embed_dim * 3, bias=False)
+        self.out_proj = nn.Linear(self.embed_dim, self.embed_dim, bias=False)
+        self.rotary_dim = None
+        if config.rotary_dim is not None:
+            self.rotary_dim = config.rotary_dim
+    def _split_heads(self, x, n_head, dim_head, mp_num):
+        reshaped = x.reshape(x.shape[:-1] + (n_head // mp_num, dim_head))
+        reshaped = reshaped.reshape(x.shape[:-2] + (-1,) + reshaped.shape[-1:])
+        return reshaped
+    def _merge_heads(self, tensor, num_attention_heads, attn_head_size):
+        """
+        Merges attn_head_size dim and num_attn_heads dim into n_ctx
+        """
+        if len(tensor.shape) == 5:
+            tensor = tensor.permute(0, 1, 3, 2, 4).contiguous()
+        elif len(tensor.shape) == 4:
+            tensor = tensor.permute(0, 2, 1, 3).contiguous()
+        else:
+            raise ValueError(f"Input tensor rank should be one of [4, 5], but is: {len(tensor.shape)}")
+        new_shape = tensor.size()[:-2] + (num_attention_heads * attn_head_size,)
+        return tensor.view(new_shape)
+    def _attn(
+        self,
+        query,
+        key,
+        value,
+        attention_mask=None,
+        head_mask=None,
+    ):
+        # compute causal mask from causal mask buffer
+        query_length, key_length = query.size(-2), key.size(-2)
+        causal_mask = self.causal_mask[:, :, key_length - query_length : key_length, :key_length]
+        # Keep the attention weights computation in fp32 to avoid overflow issues
+        query = query.to(torch.float32)
+        key = key.to(torch.float32)
+        attn_weights = torch.matmul(query, key.transpose(-1, -2))
+        attn_weights = attn_weights / self.scale_attn
+        mask_value = torch.finfo(attn_weights.dtype).min
+        # Need to be a tensor, otherwise we get error: `RuntimeError: expected scalar type float but found double`.
+        # Need to be on the same device, otherwise `RuntimeError: ..., x and y to be on the same device`
+        mask_value = torch.tensor(mask_value, dtype=attn_weights.dtype).to(attn_weights.device)
+        attn_weights = torch.where(causal_mask, attn_weights, mask_value)
+        if attention_mask is not None:
+            # Apply the attention mask
+            attn_weights = attn_weights + attention_mask
+        attn_weights = nn.Softmax(dim=-1)(attn_weights)
+        attn_weights = attn_weights.to(value.dtype)
+        attn_weights = self.attn_dropout(attn_weights)
+        # Mask heads if we want to
+        if head_mask is not None:
+            attn_weights = attn_weights * head_mask
+        attn_output = torch.matmul(attn_weights, value)
+        return attn_output, attn_weights
+    def forward(
+        self,
+        hidden_states: Optional[torch.FloatTensor],
+        attention_mask: Optional[torch.FloatTensor] = None,
+        layer_past: Optional[Tuple[torch.Tensor]] = None,
+        head_mask: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = False,
+        output_attentions: Optional[bool] = False,
+    ) -> Union[
+        Tuple[torch.Tensor, Tuple[torch.Tensor]],
+        Optional[Tuple[torch.Tensor, Tuple[torch.Tensor], Tuple[torch.Tensor, ...]]],
+    ]:
+        qkv = self.qkv_proj(hidden_states)
+        # TPU-v3
+        mp_num = 8
+        qkv_split = qkv.reshape(qkv.shape[:-1] + (mp_num, -1))
+        local_dim = self.head_dim * self.num_attention_heads // mp_num
+        query, value, key = torch.split(qkv_split, local_dim, dim=-1)
+        query = self._split_heads(query, self.num_attention_heads, self.head_dim, mp_num=mp_num)
+        key = self._split_heads(key, self.num_attention_heads, self.head_dim, mp_num=mp_num)
+        value = self._split_heads(value, self.num_attention_heads, self.head_dim, mp_num=mp_num)
+        value = value.permute(0, 2, 1, 3)
+        seq_len = key.shape[1]
+        offset = 0
+        if layer_past is not None:
+            offset = layer_past[0].shape[-2]
+            seq_len += offset
+        if self.rotary_dim is not None:
+            k_rot = key[:, :, :, : self.rotary_dim]
+            k_pass = key[:, :, :, self.rotary_dim :]
+            q_rot = query[:, :, :, : self.rotary_dim]
+            q_pass = query[:, :, :, self.rotary_dim :]
+            sincos = fixed_pos_embedding(k_rot, 1, seq_len=seq_len)
+            k_rot = apply_rotary_pos_emb(k_rot, sincos, offset=offset)
+            q_rot = apply_rotary_pos_emb(q_rot, sincos, offset=offset)
+            key = torch.cat([k_rot, k_pass], dim=-1)
+            query = torch.cat([q_rot, q_pass], dim=-1)
+        else:
+            sincos = fixed_pos_embedding(key, 1, seq_len=seq_len)
+            key = apply_rotary_pos_emb(key, sincos, offset=offset)
+            query = apply_rotary_pos_emb(query, sincos, offset=offset)
+        key = key.permute(0, 2, 1, 3)
+        query = query.permute(0, 2, 1, 3)
+        if layer_past is not None:
+            past_key = layer_past[0]
+            past_value = layer_past[1]
+            key = torch.cat((past_key, key), dim=-2)
+            value = torch.cat((past_value, value), dim=-2)
+        if use_cache is True:
+            present = (key, value)
+        else:
+            present = None
+        # compute self-attention: V x Softmax(QK^T)
+        attn_output, attn_weights = self._attn(query, key, value, attention_mask, head_mask)
+        attn_output = self._merge_heads(attn_output, self.num_attention_heads, self.head_dim)
+        attn_output = self.out_proj(attn_output)
+        attn_output = self.resid_dropout(attn_output)
+        outputs = (attn_output, present)
+        if output_attentions:
+            outputs += (attn_weights,)
+        return outputs  # a, present, (attentions)
+# Copied from transformers.models.gptj.modeling_gptj.GPTJMLP with GPTJ->CodeGen
+class CodeGenMLP(nn.Module):
+    def __init__(self, intermediate_size, config):  # in MLP: intermediate_size= 4 * embed_dim
+        super().__init__()
+        embed_dim = config.n_embd
+        self.fc_in = nn.Linear(embed_dim, intermediate_size)
+        self.fc_out = nn.Linear(intermediate_size, embed_dim)
+        self.act = ACT2FN[config.activation_function]
+        self.dropout = nn.Dropout(config.resid_pdrop)
+    def forward(self, hidden_states: Optional[torch.FloatTensor]) -> torch.FloatTensor:
+        hidden_states = self.fc_in(hidden_states)
+        hidden_states = self.act(hidden_states)
+        hidden_states = self.fc_out(hidden_states)
+        hidden_states = self.dropout(hidden_states)
+        return hidden_states
+# Copied from transformers.models.gptj.modeling_gptj.GPTJBlock with GPTJ->CodeGen
+class CodeGenBlock(nn.Module):
+    def __init__(self, config):
+        super().__init__()
+        inner_dim = config.n_inner if config.n_inner is not None else 4 * config.n_embd
+        self.ln_1 = nn.LayerNorm(config.n_embd, eps=config.layer_norm_epsilon)
+        self.attn = CodeGenAttention(config)
+        self.mlp = CodeGenMLP(inner_dim, config)
+    def forward(
+        self,
+        hidden_states: Optional[torch.FloatTensor],
+        layer_past: Optional[Tuple[torch.Tensor]] = None,
+        attention_mask: Optional[torch.FloatTensor] = None,
+        head_mask: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = False,
+        output_attentions: Optional[bool] = False,
+    ) -> Union[Tuple[torch.Tensor], Optional[Tuple[torch.Tensor, Tuple[torch.FloatTensor, ...]]]]:
+        residual = hidden_states
+        hidden_states = self.ln_1(hidden_states)
+        attn_outputs = self.attn(
+            hidden_states,
+            layer_past=layer_past,
+            attention_mask=attention_mask,
+            head_mask=head_mask,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+        )
+        attn_output = attn_outputs[0]  # output_attn: a, present, (attentions)
+        outputs = attn_outputs[1:]
+        feed_forward_hidden_states = self.mlp(hidden_states)
+        hidden_states = attn_output + feed_forward_hidden_states + residual
+        if use_cache:
+            outputs = (hidden_states,) + outputs
+        else:
+            outputs = (hidden_states,) + outputs[1:]
+        return outputs  # hidden_states, present, (attentions)
+class CodeGenPreTrainedModel(PreTrainedModel):
+    """
+    An abstract class to handle weights initialization and a simple interface for downloading and loading pretrained
+    models.
+    """
+    config_class = CodeGenConfig
+    base_model_prefix = "transformer"
+    supports_gradient_checkpointing = True
+    _no_split_modules = ["CodeGenBlock"]
+    def __init__(self, *inputs, **kwargs):
+        super().__init__(*inputs, **kwargs)
+    def _init_weights(self, module):
+        """Initialize the weights."""
+        if isinstance(module, (nn.Linear,)):
+            # Slightly different from Mesh Transformer JAX which uses truncated_normal for initialization
+            # cf https://github.com/pytorch/pytorch/pull/5617
+            module.weight.data.normal_(mean=0.0, std=self.config.initializer_range)
+            if module.bias is not None:
+                module.bias.data.zero_()
+        elif isinstance(module, nn.Embedding):
+            module.weight.data.normal_(mean=0.0, std=self.config.initializer_range)
+            if module.padding_idx is not None:
+                module.weight.data[module.padding_idx].zero_()
+        elif isinstance(module, nn.LayerNorm):
+            module.bias.data.zero_()
+            module.weight.data.fill_(1.0)
+    def _set_gradient_checkpointing(self, module, value=False):
+        if isinstance(module, CodeGenModel):
+            module.gradient_checkpointing = value
+CODEGEN_START_DOCSTRING = r"""
+    This model is a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) sub-class. Use
+    it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage and
+    behavior.
+    Parameters:
+        config ([`CodeGenConfig`]): Model configuration class with all the parameters of the model.
+            Initializing with a config file does not load the weights associated with the model, only the
+            configuration. Check out the [`~PreTrainedModel.from_pretrained`] method to load the model weights.
+"""
+CODEGEN_INPUTS_DOCSTRING = r"""
+    Args:
+        input_ids (`torch.LongTensor` of shape `({0})`):
+            Indices of input sequence tokens in the vocabulary.
+            Indices can be obtained using [`GPT2Tokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+            [What are input IDs?](../glossary#input-ids)
+        attention_mask (`torch.FloatTensor` of shape `({0})`, *optional*):
+            Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+            - 1 for tokens that are **not masked**,
+            - 0 for tokens that are **masked**.
+            [What are attention masks?](../glossary#attention-mask)
+        token_type_ids (`torch.LongTensor` of shape `({0})`, *optional*):
+            Segment token indices to indicate first and second portions of the inputs. Indices are selected in `[0,
+            1]`:
+            - 0 corresponds to a *sentence A* token,
+            - 1 corresponds to a *sentence B* token.
+            [What are token type IDs?](../glossary#token-type-ids)
+        position_ids (`torch.LongTensor` of shape `({0})`, *optional*):
+            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
+            config.n_positions - 1]`.
+            [What are position IDs?](../glossary#position-ids)
+        head_mask (`torch.FloatTensor` of shape `(num_attention_heads,)` or `(n_layer, num_attention_heads)`, *optional*):
+            Mask to nullify selected heads of the self-attention modules. Mask values selected in `[0, 1]`:
+            - 1 indicates the head is **not masked**,
+            - 0 indicates the head is **masked**.
+        inputs_embeds (`torch.FloatTensor` of shape `({0}, hidden_dim)`, *optional*):
+            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
+            is useful if you want more control over how to convert *input_ids* indices into associated vectors than the
+            model's internal embedding lookup matrix.
+        output_attentions (`bool`, *optional*):
+            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
+            tensors for more detail.
+        output_hidden_states (`bool`, *optional*):
+            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
+            more detail.
+        return_dict (`bool`, *optional*):
+            Whether or not to return a [`~utils.ModelOutput`] instead of a plain tuple.
+"""
+@add_start_docstrings(
+    "The bare CodeGen Model transformer outputting raw hidden-states without any specific head on top.",
+    CODEGEN_START_DOCSTRING,
+)
+class CodeGenModel(CodeGenPreTrainedModel):
+    def __init__(self, config):
+        super().__init__(config)
+        self.embed_dim = config.n_embd
+        self.vocab_size = config.vocab_size
+        self.wte = nn.Embedding(config.vocab_size, self.embed_dim)
+        self.drop = nn.Dropout(config.embd_pdrop)
+        self.h = nn.ModuleList([CodeGenBlock(config) for _ in range(config.n_layer)])
+        self.ln_f = nn.LayerNorm(self.embed_dim, eps=config.layer_norm_epsilon)
+        self.rotary_dim = min(config.rotary_dim, config.n_ctx // config.num_attention_heads)
+        self.gradient_checkpointing = False
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_input_embeddings(self):
+        return self.wte
+    def set_input_embeddings(self, new_embeddings):
+        self.wte = new_embeddings
+    @add_start_docstrings_to_model_forward(CODEGEN_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
+    @add_code_sample_docstrings(
+        processor_class=_TOKENIZER_FOR_DOC,
+        checkpoint=_CHECKPOINT_FOR_DOC,
+        output_type=BaseModelOutputWithPast,
+        config_class=_CONFIG_FOR_DOC,
+    )
+    def forward(
+        self,
+        input_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[Tuple[Tuple[torch.Tensor]]] = None,
+        attention_mask: Optional[torch.FloatTensor] = None,
+        token_type_ids: Optional[torch.LongTensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        head_mask: Optional[torch.FloatTensor] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, BaseModelOutputWithPast]:
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if input_ids is not None and inputs_embeds is not None:
+            raise ValueError("You cannot specify both input_ids and inputs_embeds at the same time")
+        elif input_ids is not None:
+            input_shape = input_ids.size()
+            input_ids = input_ids.view(-1, input_shape[-1])
+            batch_size = input_ids.shape[0]
+        elif inputs_embeds is not None:
+            input_shape = inputs_embeds.size()[:-1]
+            batch_size = inputs_embeds.shape[0]
+        else:
+            raise ValueError("You have to specify either input_ids or inputs_embeds")
+        device = input_ids.device if input_ids is not None else inputs_embeds.device
+        if token_type_ids is not None:
+            token_type_ids = token_type_ids.view(-1, input_shape[-1])
+        if position_ids is not None:
+            position_ids = position_ids.view(-1, input_shape[-1])
+        if past_key_values is None:
+            past_length = 0
+            past_key_values = tuple([None] * len(self.h))
+        else:
+            past_length = past_key_values[0][0].size(-2)
+        if position_ids is None:
+            position_ids = torch.arange(past_length, input_shape[-1] + past_length, dtype=torch.long, device=device)
+            position_ids = position_ids.unsqueeze(0).view(-1, input_shape[-1])
+        # Attention mask.
+        if attention_mask is not None:
+            if batch_size <= 0:
+                raise ValueError("batch_size has to be defined and > 0")
+            attention_mask = attention_mask.view(batch_size, -1)
+            # We create a 3D attention mask from a 2D tensor mask.
+            # Sizes are [batch_size, 1, 1, to_seq_length]
+            # So we can broadcast to [batch_size, num_heads, from_seq_length, to_seq_length]
+            # this attention mask is more simple than the triangular masking of causal attention
+            # used in OpenAI GPT, we just need to prepare the broadcast dimension here.
+            attention_mask = attention_mask[:, None, None, :]
+            # Since attention_mask is 1.0 for positions we want to attend and 0.0 for
+            # masked positions, this operation will create a tensor which is 0.0 for
+            # positions we want to attend and the dtype's smallest value for masked positions.
+            # Since we are adding it to the raw scores before the softmax, this is
+            # effectively the same as removing these entirely.
+            attention_mask = attention_mask.to(dtype=self.dtype)  # fp16 compatibility
+            attention_mask = (1.0 - attention_mask) * torch.finfo(self.dtype).min
+        # Prepare head mask if needed
+        # 1.0 in head_mask indicate we keep the head
+        # attention_probs has shape bsz x num_attention_heads x N x N
+        # head_mask has shape n_layer x batch x num_attention_heads x N x N
+        head_mask = self.get_head_mask(head_mask, self.config.n_layer)
+        if inputs_embeds is None:
+            inputs_embeds = self.wte(input_ids)
+        hidden_states = inputs_embeds
+        if token_type_ids is not None:
+            token_type_embeds = self.wte(token_type_ids)
+            hidden_states = hidden_states + token_type_embeds
+        hidden_states = self.drop(hidden_states)
+        output_shape = input_shape + (hidden_states.size(-1),)
+        presents = () if use_cache else None
+        all_self_attentions = () if output_attentions else None
+        all_hidden_states = () if output_hidden_states else None
+        for i, (block, layer_past) in enumerate(zip(self.h, past_key_values)):
+            if output_hidden_states:
+                all_hidden_states = all_hidden_states + (hidden_states,)
+            if self.gradient_checkpointing and self.training:
+                if use_cache:
+                    logger.warning(
+                        "`use_cache=True` is incompatible with `config.gradient_checkpointing=True`. Setting "
+                        "`use_cache=False`..."
+                    )
+                    use_cache = False
+                def create_custom_forward(module):
+                    def custom_forward(*inputs):
+                        # None for past_key_value
+                        return module(*inputs, use_cache, output_attentions)
+                    return custom_forward
+                outputs = torch.utils.checkpoint.checkpoint(
+                    create_custom_forward(block),
+                    hidden_states,
+                    None,
+                    attention_mask,
+                    head_mask[i],
+                )
+            else:
+                outputs = block(
+                    hidden_states,
+                    layer_past=layer_past,
+                    attention_mask=attention_mask,
+                    head_mask=head_mask[i],
+                    use_cache=use_cache,
+                    output_attentions=output_attentions,
+                )
+            hidden_states = outputs[0]
+            if use_cache is True:
+                presents = presents + (outputs[1],)
+            if output_attentions:
+                all_self_attentions = all_self_attentions + (outputs[2 if use_cache else 1],)
+        hidden_states = self.ln_f(hidden_states)
+        hidden_states = hidden_states.view(output_shape)
+        # Add last hidden state
+        if output_hidden_states:
+            all_hidden_states = all_hidden_states + (hidden_states,)
+        if not return_dict:
+            return tuple(v for v in [hidden_states, presents, all_hidden_states, all_self_attentions] if v is not None)
+        return BaseModelOutputWithPast(
+            last_hidden_state=hidden_states,
+            past_key_values=presents,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attentions,
+        )
+@add_start_docstrings(
+    """
+    The CodeGen Model transformer with a language modeling head on top.
+    """,
+    CODEGEN_START_DOCSTRING,
+)
+class CodeGenForCausalLM(CodeGenPreTrainedModel):
+    _keys_to_ignore_on_load_missing = [r"h\.\d+\.attn\.masked_bias", r"h\.\d+\.attn\.bias"]
+    def __init__(self, config):
+        super().__init__(config)
+        self.transformer = CodeGenModel(config)
+        self.lm_head = nn.Linear(config.n_embd, config.vocab_size)
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_output_embeddings(self):
+        return self.lm_head
+    def set_output_embeddings(self, new_embeddings):
+        self.lm_head = new_embeddings
+    def prepare_inputs_for_generation(self, input_ids, past=None, **kwargs):
+        token_type_ids = kwargs.get("token_type_ids", None)
+        # only last token for inputs_ids if past is defined in kwargs
+        if past:
+            input_ids = input_ids[:, -1].unsqueeze(-1)
+            if token_type_ids is not None:
+                token_type_ids = token_type_ids[:, -1].unsqueeze(-1)
+        attention_mask = kwargs.get("attention_mask", None)
+        position_ids = kwargs.get("position_ids", None)
+        if attention_mask is not None and position_ids is None:
+            # create position_ids on the fly for batch generation
+            position_ids = attention_mask.long().cumsum(-1) - 1
+            position_ids.masked_fill_(attention_mask == 0, 1)
+            if past:
+                position_ids = position_ids[:, -1].unsqueeze(-1)
+        else:
+            position_ids = None
+        return {
+            "input_ids": input_ids,
+            "past_key_values": past,
+            "use_cache": kwargs.get("use_cache"),
+            "position_ids": position_ids,
+            "attention_mask": attention_mask,
+            "token_type_ids": token_type_ids,
+        }
+    @add_start_docstrings_to_model_forward(CODEGEN_INPUTS_DOCSTRING.format("batch_size, sequence_length"))
+    @add_code_sample_docstrings(
+        processor_class=_TOKENIZER_FOR_DOC,
+        checkpoint=_CHECKPOINT_FOR_DOC,
+        output_type=CausalLMOutputWithPast,
+        config_class=_CONFIG_FOR_DOC,
+    )
+    def forward(
+        self,
+        input_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[Tuple[Tuple[torch.Tensor]]] = None,
+        attention_mask: Optional[torch.FloatTensor] = None,
+        token_type_ids: Optional[torch.LongTensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        head_mask: Optional[torch.FloatTensor] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        r"""
+        labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Labels for language modeling. Note that the labels **are shifted** inside the model, i.e. you can set
+            `labels = input_ids` Indices are selected in `[-100, 0, ..., config.vocab_size]` All labels set to `-100`
+            are ignored (masked), the loss is only computed for labels in `[0, ..., config.vocab_size]`
+        """
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        transformer_outputs = self.transformer(
+            input_ids,
+            past_key_values=past_key_values,
+            attention_mask=attention_mask,
+            token_type_ids=token_type_ids,
+            position_ids=position_ids,
+            head_mask=head_mask,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+        )
+        hidden_states = transformer_outputs[0]
+        # make sure sampling in fp16 works correctly and
+        # compute loss in fp32 to match with mesh-tf version
+        # https://github.com/EleutherAI/gpt-neo/blob/89ce74164da2fb16179106f54e2269b5da8db333/models/gpt2/gpt2.py#L179
+        lm_logits = self.lm_head(hidden_states).to(torch.float32)
+        loss = None
+        if labels is not None:
+            # Shift so that tokens < n predict n
+            shift_logits = lm_logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            # Flatten the tokens
+            loss_fct = CrossEntropyLoss()
+            loss = loss_fct(shift_logits.view(-1, shift_logits.size(-1)), shift_labels.view(-1))
+            loss = loss.to(hidden_states.dtype)
+        if not return_dict:
+            output = (lm_logits,) + transformer_outputs[1:]
+            return ((loss,) + output) if loss is not None else output
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=lm_logits,
+            past_key_values=transformer_outputs.past_key_values,
+            hidden_states=transformer_outputs.hidden_states,
+            attentions=transformer_outputs.attentions,
+        )
+    @staticmethod
+    def _reorder_cache(past: Tuple[Tuple[torch.Tensor]], beam_idx: torch.Tensor) -> Tuple[Tuple[torch.Tensor]]:
+        """
+        This function is used to re-order the `past_key_values` cache if [`~PretrainedModel.beam_search`] or
+        [`~PretrainedModel.beam_sample`] is called. This is required to match `past_key_values` with the correct
+        beam_idx at every generation step.
+        """
+        return tuple(
+            tuple(past_state.index_select(0, beam_idx.to(past_state.device)) for past_state in layer_past)
+            for layer_past in past
+        )

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "bos_token": "<|endoftext|>",
+  "eos_token": "<|endoftext|>",
+  "unk_token": "<|endoftext|>"
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "add_prefix_space": false,
+  "bos_token": "<|endoftext|>",
+  "eos_token": "<|endoftext|>",
+  "model_max_length": 1024,
+  "name_or_path": "gpt2",
+  "special_tokens_map_file": null,
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff